)]}'
{"id":"LineageOS%2Fandroid_kernel_oneplus_msm8998~323455","triplet_id":"LineageOS%2Fandroid_kernel_oneplus_msm8998~lineage-19.0~Iab6c63ba26730279d2699057ea1747d52e2a20fa","project":"LineageOS/android_kernel_oneplus_msm8998","branch":"lineage-19.0","full_branch":"refs/heads/lineage-19.0","topic":"twelve-op8998","attention_set":{},"removed_from_attention_set":{"13648":{"account":{"_account_id":13648,"name":"Bruno Martins","email":"bgcngm@gmail.com","username":"bgcngm","avatars":[{"url":"https://www.gravatar.com/avatar/3d939ee28d51d14e76de3a4510b309ce.jpg?d\u003didenticon\u0026r\u003dpg\u0026s\u003d32","height":32},{"url":"https://www.gravatar.com/avatar/3d939ee28d51d14e76de3a4510b309ce.jpg?d\u003didenticon\u0026r\u003dpg\u0026s\u003d56","height":56},{"url":"https://www.gravatar.com/avatar/3d939ee28d51d14e76de3a4510b309ce.jpg?d\u003didenticon\u0026r\u003dpg\u0026s\u003d100","height":100},{"url":"https://www.gravatar.com/avatar/3d939ee28d51d14e76de3a4510b309ce.jpg?d\u003didenticon\u0026r\u003dpg\u0026s\u003d120","height":120}]},"last_update":"2022-03-14 10:53:36.000000000","reason":"Change was abandoned"},"5911":{"account":{"_account_id":5911,"name":"Georg Veichtlbauer","email":"georg@vware.at","username":"veichtlbauer","avatars":[{"url":"https://www.gravatar.com/avatar/39db6f16bd92d063f8a1762ba2009d16.jpg?d\u003didenticon\u0026r\u003dpg\u0026s\u003d32","height":32},{"url":"https://www.gravatar.com/avatar/39db6f16bd92d063f8a1762ba2009d16.jpg?d\u003didenticon\u0026r\u003dpg\u0026s\u003d56","height":56},{"url":"https://www.gravatar.com/avatar/39db6f16bd92d063f8a1762ba2009d16.jpg?d\u003didenticon\u0026r\u003dpg\u0026s\u003d100","height":100},{"url":"https://www.gravatar.com/avatar/39db6f16bd92d063f8a1762ba2009d16.jpg?d\u003didenticon\u0026r\u003dpg\u0026s\u003d120","height":120}]},"last_update":"2022-02-12 06:15:33.000000000","reason":"removed on reply"}},"hashtags":[],"change_id":"Iab6c63ba26730279d2699057ea1747d52e2a20fa","subject":"Backport BPF","status":"ABANDONED","created":"2022-02-11 21:36:42.000000000","updated":"2022-03-14 10:53:36.000000000","total_comment_count":9,"unresolved_comment_count":0,"has_review_started":true,"meta_rev_id":"ae05d2ca968eecb9847ec0c8a2849c3032470910","_number":323455,"virtual_id_number":323455,"owner":{"_account_id":5911,"name":"Georg Veichtlbauer","email":"georg@vware.at","username":"veichtlbauer","avatars":[{"url":"https://www.gravatar.com/avatar/39db6f16bd92d063f8a1762ba2009d16.jpg?d\u003didenticon\u0026r\u003dpg\u0026s\u003d32","height":32},{"url":"https://www.gravatar.com/avatar/39db6f16bd92d063f8a1762ba2009d16.jpg?d\u003didenticon\u0026r\u003dpg\u0026s\u003d56","height":56},{"url":"https://www.gravatar.com/avatar/39db6f16bd92d063f8a1762ba2009d16.jpg?d\u003didenticon\u0026r\u003dpg\u0026s\u003d100","height":100},{"url":"https://www.gravatar.com/avatar/39db6f16bd92d063f8a1762ba2009d16.jpg?d\u003didenticon\u0026r\u003dpg\u0026s\u003d120","height":120}]},"actions":{},"labels":{"Verified":{"values":{"-1":"Fails"," 0":"No score","+1":"Verified"},"description":"","default_value":0},"Code-Review":{"values":{"-2":"Do not submit","-1":"I would prefer that you didn\u0027t submit this"," 0":"No score","+1":"Looks good to me, but someone else must approve","+2":"Looks good to me, approved"},"description":"","default_value":0},"CI":{"values":{"-1":"Fail"," 0":"No score","+1":"Pass"},"description":"","default_value":0,"optional":true}},"removable_reviewers":[],"reviewers":{"CC":[{"_account_id":13648,"name":"Bruno Martins","email":"bgcngm@gmail.com","username":"bgcngm","avatars":[{"url":"https://www.gravatar.com/avatar/3d939ee28d51d14e76de3a4510b309ce.jpg?d\u003didenticon\u0026r\u003dpg\u0026s\u003d32","height":32},{"url":"https://www.gravatar.com/avatar/3d939ee28d51d14e76de3a4510b309ce.jpg?d\u003didenticon\u0026r\u003dpg\u0026s\u003d56","height":56},{"url":"https://www.gravatar.com/avatar/3d939ee28d51d14e76de3a4510b309ce.jpg?d\u003didenticon\u0026r\u003dpg\u0026s\u003d100","height":100},{"url":"https://www.gravatar.com/avatar/3d939ee28d51d14e76de3a4510b309ce.jpg?d\u003didenticon\u0026r\u003dpg\u0026s\u003d120","height":120}]}]},"pending_reviewers":{},"reviewer_updates":[{"updated":"2022-02-11 22:37:39.000000000","updated_by":{"_account_id":13648,"name":"Bruno Martins","email":"bgcngm@gmail.com","username":"bgcngm","avatars":[{"url":"https://www.gravatar.com/avatar/3d939ee28d51d14e76de3a4510b309ce.jpg?d\u003didenticon\u0026r\u003dpg\u0026s\u003d32","height":32},{"url":"https://www.gravatar.com/avatar/3d939ee28d51d14e76de3a4510b309ce.jpg?d\u003didenticon\u0026r\u003dpg\u0026s\u003d56","height":56},{"url":"https://www.gravatar.com/avatar/3d939ee28d51d14e76de3a4510b309ce.jpg?d\u003didenticon\u0026r\u003dpg\u0026s\u003d100","height":100},{"url":"https://www.gravatar.com/avatar/3d939ee28d51d14e76de3a4510b309ce.jpg?d\u003didenticon\u0026r\u003dpg\u0026s\u003d120","height":120}]},"real_updated_by":{"_account_id":13648,"name":"Bruno Martins","email":"bgcngm@gmail.com","username":"bgcngm","avatars":[{"url":"https://www.gravatar.com/avatar/3d939ee28d51d14e76de3a4510b309ce.jpg?d\u003didenticon\u0026r\u003dpg\u0026s\u003d32","height":32},{"url":"https://www.gravatar.com/avatar/3d939ee28d51d14e76de3a4510b309ce.jpg?d\u003didenticon\u0026r\u003dpg\u0026s\u003d56","height":56},{"url":"https://www.gravatar.com/avatar/3d939ee28d51d14e76de3a4510b309ce.jpg?d\u003didenticon\u0026r\u003dpg\u0026s\u003d100","height":100},{"url":"https://www.gravatar.com/avatar/3d939ee28d51d14e76de3a4510b309ce.jpg?d\u003didenticon\u0026r\u003dpg\u0026s\u003d120","height":120}]},"reviewer":{"_account_id":13648,"name":"Bruno Martins","email":"bgcngm@gmail.com","username":"bgcngm","avatars":[{"url":"https://www.gravatar.com/avatar/3d939ee28d51d14e76de3a4510b309ce.jpg?d\u003didenticon\u0026r\u003dpg\u0026s\u003d32","height":32},{"url":"https://www.gravatar.com/avatar/3d939ee28d51d14e76de3a4510b309ce.jpg?d\u003didenticon\u0026r\u003dpg\u0026s\u003d56","height":56},{"url":"https://www.gravatar.com/avatar/3d939ee28d51d14e76de3a4510b309ce.jpg?d\u003didenticon\u0026r\u003dpg\u0026s\u003d100","height":100},{"url":"https://www.gravatar.com/avatar/3d939ee28d51d14e76de3a4510b309ce.jpg?d\u003didenticon\u0026r\u003dpg\u0026s\u003d120","height":120}]},"state":"CC"}],"messages":[{"id":"efc24a5380fb81d17445a8efd18f521028bb603b","tag":"autogenerated:gerrit:newPatchSet","author":{"_account_id":5911,"name":"Georg Veichtlbauer","email":"georg@vware.at","username":"veichtlbauer","avatars":[{"url":"https://www.gravatar.com/avatar/39db6f16bd92d063f8a1762ba2009d16.jpg?d\u003didenticon\u0026r\u003dpg\u0026s\u003d32","height":32},{"url":"https://www.gravatar.com/avatar/39db6f16bd92d063f8a1762ba2009d16.jpg?d\u003didenticon\u0026r\u003dpg\u0026s\u003d56","height":56},{"url":"https://www.gravatar.com/avatar/39db6f16bd92d063f8a1762ba2009d16.jpg?d\u003didenticon\u0026r\u003dpg\u0026s\u003d100","height":100},{"url":"https://www.gravatar.com/avatar/39db6f16bd92d063f8a1762ba2009d16.jpg?d\u003didenticon\u0026r\u003dpg\u0026s\u003d120","height":120}]},"date":"2022-02-11 21:36:42.000000000","message":"Uploaded patch set 1.","accounts_in_message":[],"_revision_number":1},{"id":"f47434b10fa670001c5ad1abcc830b0849b45061","author":{"_account_id":13648,"name":"Bruno Martins","email":"bgcngm@gmail.com","username":"bgcngm","avatars":[{"url":"https://www.gravatar.com/avatar/3d939ee28d51d14e76de3a4510b309ce.jpg?d\u003didenticon\u0026r\u003dpg\u0026s\u003d32","height":32},{"url":"https://www.gravatar.com/avatar/3d939ee28d51d14e76de3a4510b309ce.jpg?d\u003didenticon\u0026r\u003dpg\u0026s\u003d56","height":56},{"url":"https://www.gravatar.com/avatar/3d939ee28d51d14e76de3a4510b309ce.jpg?d\u003didenticon\u0026r\u003dpg\u0026s\u003d100","height":100},{"url":"https://www.gravatar.com/avatar/3d939ee28d51d14e76de3a4510b309ce.jpg?d\u003didenticon\u0026r\u003dpg\u0026s\u003d120","height":120}]},"date":"2022-02-11 22:37:39.000000000","message":"Patch Set 1:\n\n(1 comment)","accounts_in_message":[],"_revision_number":1},{"id":"ffee07dd57781f72668a0ff62e9df7c6cf5fda48","author":{"_account_id":13648,"name":"Bruno Martins","email":"bgcngm@gmail.com","username":"bgcngm","avatars":[{"url":"https://www.gravatar.com/avatar/3d939ee28d51d14e76de3a4510b309ce.jpg?d\u003didenticon\u0026r\u003dpg\u0026s\u003d32","height":32},{"url":"https://www.gravatar.com/avatar/3d939ee28d51d14e76de3a4510b309ce.jpg?d\u003didenticon\u0026r\u003dpg\u0026s\u003d56","height":56},{"url":"https://www.gravatar.com/avatar/3d939ee28d51d14e76de3a4510b309ce.jpg?d\u003didenticon\u0026r\u003dpg\u0026s\u003d100","height":100},{"url":"https://www.gravatar.com/avatar/3d939ee28d51d14e76de3a4510b309ce.jpg?d\u003didenticon\u0026r\u003dpg\u0026s\u003d120","height":120}]},"date":"2022-02-11 23:14:26.000000000","message":"Patch Set 1:\n\n(1 comment)","accounts_in_message":[],"_revision_number":1},{"id":"90eef5c4a7f2e7a03d665f378dae5989a201b2cd","author":{"_account_id":13648,"name":"Bruno Martins","email":"bgcngm@gmail.com","username":"bgcngm","avatars":[{"url":"https://www.gravatar.com/avatar/3d939ee28d51d14e76de3a4510b309ce.jpg?d\u003didenticon\u0026r\u003dpg\u0026s\u003d32","height":32},{"url":"https://www.gravatar.com/avatar/3d939ee28d51d14e76de3a4510b309ce.jpg?d\u003didenticon\u0026r\u003dpg\u0026s\u003d56","height":56},{"url":"https://www.gravatar.com/avatar/3d939ee28d51d14e76de3a4510b309ce.jpg?d\u003didenticon\u0026r\u003dpg\u0026s\u003d100","height":100},{"url":"https://www.gravatar.com/avatar/3d939ee28d51d14e76de3a4510b309ce.jpg?d\u003didenticon\u0026r\u003dpg\u0026s\u003d120","height":120}]},"date":"2022-02-11 23:28:22.000000000","message":"Patch Set 1:\n\n(1 comment)","accounts_in_message":[],"_revision_number":1},{"id":"52580057f4244fe944c55d80525e9a8b8872c5f7","author":{"_account_id":5911,"name":"Georg Veichtlbauer","email":"georg@vware.at","username":"veichtlbauer","avatars":[{"url":"https://www.gravatar.com/avatar/39db6f16bd92d063f8a1762ba2009d16.jpg?d\u003didenticon\u0026r\u003dpg\u0026s\u003d32","height":32},{"url":"https://www.gravatar.com/avatar/39db6f16bd92d063f8a1762ba2009d16.jpg?d\u003didenticon\u0026r\u003dpg\u0026s\u003d56","height":56},{"url":"https://www.gravatar.com/avatar/39db6f16bd92d063f8a1762ba2009d16.jpg?d\u003didenticon\u0026r\u003dpg\u0026s\u003d100","height":100},{"url":"https://www.gravatar.com/avatar/39db6f16bd92d063f8a1762ba2009d16.jpg?d\u003didenticon\u0026r\u003dpg\u0026s\u003d120","height":120}]},"date":"2022-02-12 06:15:33.000000000","message":"Patch Set 1:\n\n(3 comments)","accounts_in_message":[],"_revision_number":1},{"id":"2df2b89abdfb75602e5c660e82955d708a14e832","tag":"autogenerated:gerrit:newPatchSet","author":{"_account_id":5911,"name":"Georg Veichtlbauer","email":"georg@vware.at","username":"veichtlbauer","avatars":[{"url":"https://www.gravatar.com/avatar/39db6f16bd92d063f8a1762ba2009d16.jpg?d\u003didenticon\u0026r\u003dpg\u0026s\u003d32","height":32},{"url":"https://www.gravatar.com/avatar/39db6f16bd92d063f8a1762ba2009d16.jpg?d\u003didenticon\u0026r\u003dpg\u0026s\u003d56","height":56},{"url":"https://www.gravatar.com/avatar/39db6f16bd92d063f8a1762ba2009d16.jpg?d\u003didenticon\u0026r\u003dpg\u0026s\u003d100","height":100},{"url":"https://www.gravatar.com/avatar/39db6f16bd92d063f8a1762ba2009d16.jpg?d\u003didenticon\u0026r\u003dpg\u0026s\u003d120","height":120}]},"date":"2022-02-12 06:17:58.000000000","message":"Uploaded patch set 2.","accounts_in_message":[],"_revision_number":2},{"id":"64b27ec8a7457a6d1e0e9a223b7fadf7b8712642","tag":"autogenerated:gerrit:setTopic","author":{"_account_id":5911,"name":"Georg Veichtlbauer","email":"georg@vware.at","username":"veichtlbauer","avatars":[{"url":"https://www.gravatar.com/avatar/39db6f16bd92d063f8a1762ba2009d16.jpg?d\u003didenticon\u0026r\u003dpg\u0026s\u003d32","height":32},{"url":"https://www.gravatar.com/avatar/39db6f16bd92d063f8a1762ba2009d16.jpg?d\u003didenticon\u0026r\u003dpg\u0026s\u003d56","height":56},{"url":"https://www.gravatar.com/avatar/39db6f16bd92d063f8a1762ba2009d16.jpg?d\u003didenticon\u0026r\u003dpg\u0026s\u003d100","height":100},{"url":"https://www.gravatar.com/avatar/39db6f16bd92d063f8a1762ba2009d16.jpg?d\u003didenticon\u0026r\u003dpg\u0026s\u003d120","height":120}]},"date":"2022-02-12 06:49:20.000000000","message":"Topic set to twelve-op8998","accounts_in_message":[],"_revision_number":2},{"id":"194771defc51fced2962c77ddcbeaa29a25c56d4","author":{"_account_id":5911,"name":"Georg Veichtlbauer","email":"georg@vware.at","username":"veichtlbauer","avatars":[{"url":"https://www.gravatar.com/avatar/39db6f16bd92d063f8a1762ba2009d16.jpg?d\u003didenticon\u0026r\u003dpg\u0026s\u003d32","height":32},{"url":"https://www.gravatar.com/avatar/39db6f16bd92d063f8a1762ba2009d16.jpg?d\u003didenticon\u0026r\u003dpg\u0026s\u003d56","height":56},{"url":"https://www.gravatar.com/avatar/39db6f16bd92d063f8a1762ba2009d16.jpg?d\u003didenticon\u0026r\u003dpg\u0026s\u003d100","height":100},{"url":"https://www.gravatar.com/avatar/39db6f16bd92d063f8a1762ba2009d16.jpg?d\u003didenticon\u0026r\u003dpg\u0026s\u003d120","height":120}]},"date":"2022-02-12 07:34:22.000000000","message":"Patch Set 2:\n\n(1 comment)","accounts_in_message":[],"_revision_number":2},{"id":"8fcfbc8f8dd160b6a924ed584e0a3b69846ed574","tag":"autogenerated:gerrit:newPatchSet","author":{"_account_id":5911,"name":"Georg Veichtlbauer","email":"georg@vware.at","username":"veichtlbauer","avatars":[{"url":"https://www.gravatar.com/avatar/39db6f16bd92d063f8a1762ba2009d16.jpg?d\u003didenticon\u0026r\u003dpg\u0026s\u003d32","height":32},{"url":"https://www.gravatar.com/avatar/39db6f16bd92d063f8a1762ba2009d16.jpg?d\u003didenticon\u0026r\u003dpg\u0026s\u003d56","height":56},{"url":"https://www.gravatar.com/avatar/39db6f16bd92d063f8a1762ba2009d16.jpg?d\u003didenticon\u0026r\u003dpg\u0026s\u003d100","height":100},{"url":"https://www.gravatar.com/avatar/39db6f16bd92d063f8a1762ba2009d16.jpg?d\u003didenticon\u0026r\u003dpg\u0026s\u003d120","height":120}]},"date":"2022-02-12 08:23:26.000000000","message":"Uploaded patch set 3.","accounts_in_message":[],"_revision_number":3},{"id":"4f6539aabf27865c94c4211c8364791f1bfc28f9","author":{"_account_id":5911,"name":"Georg Veichtlbauer","email":"georg@vware.at","username":"veichtlbauer","avatars":[{"url":"https://www.gravatar.com/avatar/39db6f16bd92d063f8a1762ba2009d16.jpg?d\u003didenticon\u0026r\u003dpg\u0026s\u003d32","height":32},{"url":"https://www.gravatar.com/avatar/39db6f16bd92d063f8a1762ba2009d16.jpg?d\u003didenticon\u0026r\u003dpg\u0026s\u003d56","height":56},{"url":"https://www.gravatar.com/avatar/39db6f16bd92d063f8a1762ba2009d16.jpg?d\u003didenticon\u0026r\u003dpg\u0026s\u003d100","height":100},{"url":"https://www.gravatar.com/avatar/39db6f16bd92d063f8a1762ba2009d16.jpg?d\u003didenticon\u0026r\u003dpg\u0026s\u003d120","height":120}]},"date":"2022-02-12 08:26:15.000000000","message":"Patch Set 3:\n\n(2 comments)","accounts_in_message":[],"_revision_number":3},{"id":"ae05d2ca968eecb9847ec0c8a2849c3032470910","tag":"autogenerated:gerrit:abandon","author":{"_account_id":5911,"name":"Georg Veichtlbauer","email":"georg@vware.at","username":"veichtlbauer","avatars":[{"url":"https://www.gravatar.com/avatar/39db6f16bd92d063f8a1762ba2009d16.jpg?d\u003didenticon\u0026r\u003dpg\u0026s\u003d32","height":32},{"url":"https://www.gravatar.com/avatar/39db6f16bd92d063f8a1762ba2009d16.jpg?d\u003didenticon\u0026r\u003dpg\u0026s\u003d56","height":56},{"url":"https://www.gravatar.com/avatar/39db6f16bd92d063f8a1762ba2009d16.jpg?d\u003didenticon\u0026r\u003dpg\u0026s\u003d100","height":100},{"url":"https://www.gravatar.com/avatar/39db6f16bd92d063f8a1762ba2009d16.jpg?d\u003didenticon\u0026r\u003dpg\u0026s\u003d120","height":120}]},"date":"2022-03-14 10:53:36.000000000","message":"Abandoned\n\npushed individual commits","accounts_in_message":[],"_revision_number":3}],"current_revision_number":3,"current_revision":"c794a655dc883b90f718519667f5a91cda4e9204","revisions":{"838b49504e7c00f10a4cd9cf83d34036f4af3090":{"kind":"REWORK","_number":1,"created":"2022-02-11 21:36:42.000000000","uploader":{"_account_id":5911,"name":"Georg Veichtlbauer","email":"georg@vware.at","username":"veichtlbauer","avatars":[{"url":"https://www.gravatar.com/avatar/39db6f16bd92d063f8a1762ba2009d16.jpg?d\u003didenticon\u0026r\u003dpg\u0026s\u003d32","height":32},{"url":"https://www.gravatar.com/avatar/39db6f16bd92d063f8a1762ba2009d16.jpg?d\u003didenticon\u0026r\u003dpg\u0026s\u003d56","height":56},{"url":"https://www.gravatar.com/avatar/39db6f16bd92d063f8a1762ba2009d16.jpg?d\u003didenticon\u0026r\u003dpg\u0026s\u003d100","height":100},{"url":"https://www.gravatar.com/avatar/39db6f16bd92d063f8a1762ba2009d16.jpg?d\u003didenticon\u0026r\u003dpg\u0026s\u003d120","height":120}]},"ref":"refs/changes/55/323455/1","fetch":{"anonymous http":{"url":"https://github.com/LineageOS/android_kernel_oneplus_msm8998","ref":"refs/changes/55/323455/1","commands":{"Branch":"git fetch https://github.com/LineageOS/android_kernel_oneplus_msm8998 refs/changes/55/323455/1 \u0026\u0026 git checkout -b change-323455 FETCH_HEAD","Checkout":"git fetch https://github.com/LineageOS/android_kernel_oneplus_msm8998 refs/changes/55/323455/1 \u0026\u0026 git checkout FETCH_HEAD","Cherry Pick":"git fetch https://github.com/LineageOS/android_kernel_oneplus_msm8998 refs/changes/55/323455/1 \u0026\u0026 git cherry-pick FETCH_HEAD","Format Patch":"git fetch https://github.com/LineageOS/android_kernel_oneplus_msm8998 refs/changes/55/323455/1 \u0026\u0026 git format-patch -1 --stdout FETCH_HEAD","Pull":"git pull https://github.com/LineageOS/android_kernel_oneplus_msm8998 refs/changes/55/323455/1","Reset To":"git fetch https://github.com/LineageOS/android_kernel_oneplus_msm8998 refs/changes/55/323455/1 \u0026\u0026 git reset --hard FETCH_HEAD"}}},"commit":{"parents":[{"commit":"d3dee68c56d18f064801d121e6c734b4cb806b80","subject":"Merge branch \u0027google/android-4.4-p\u0027 into lineage-18.1","web_links":[{"name":"GitHub","tooltip":"Open in GitWeb","url":"https://github.com/LineageOS/android_kernel_oneplus_msm8998/commit/d3dee68c56d18f064801d121e6c734b4cb806b80"}]}],"author":{"name":"Georg Veichtlbauer","email":"georg@vware.at","date":"2022-02-11 21:35:17.000000000","tz":60},"committer":{"name":"Georg Veichtlbauer","email":"georg@vware.at","date":"2022-02-11 21:35:17.000000000","tz":60},"subject":"Backport BPF","message":"Backport BPF\n\nSquashed commit of the following:\n\ncommit b6b4eb461b596b7715a5dc968d4ce2e766f10510\nAuthor: ivanmeler \u003ci_ivan@windowslive.com\u003e\nDate:   Thu Dec 2 08:30:24 2021 +0000\n\n    Revert \"scripts/setlocalversion: make git describe output more reliable\"\n\n    This reverts commit 6867b79c552a39871f7b8c0499cf5b88c11da1b8.\n\ncommit 9445e4c9e34de2e77f12e0924e373121555d6bb9\nAuthor: ivanmeler \u003ci_ivan@windowslive.com\u003e\nDate:   Tue Oct 26 17:25:51 2021 +0000\n\n    oneplus5: Remove wireguard\n\n    Change-Id: Ib9d794d2fd88c9d583a1c23e3f24313546abcffc\n\ncommit bb4ab6e20ca3313cfc072b7a19e8dde6ccac396d\nAuthor: Chatur27 \u003cjasonbright2709@gmail.com\u003e\nDate:   Sun Oct 10 19:36:47 2021 +0000\n\n    treewide: Fixup for BPF backport\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n    Signed-off-by: roynatech2544 \u003cwhiteshell2544@naver.com\u003e\n\ncommit a476ee0dabab970f4e07fb2eb12c2a21cd46f53a\nAuthor: ivanmeler \u003ci_ivan@windowslive.com\u003e\nDate:   Thu Oct 28 10:00:00 2021 +0000\n\n    oneplus5: Enable BPF\n\n    Change-Id: I3695f04c93f159ed7afce7a865aafd70a24818a3\n\ncommit bb3bc0bb892b5c2f5f1a64b5311efa1ca9af8103\nAuthor: Maitreya29 \u003cMaitreyapatni30@gmail.com\u003e\nDate:   Mon Aug 30 11:35:04 2021 +0530\n\n    net: adapt bpf_xdp_copy\n\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit fef4fce8b3e985fc39813c0d5c802f8a83bcbfd0\nAuthor: Maitreya29 \u003cMaitreyapatni30@gmail.com\u003e\nDate:   Mon Aug 30 10:55:49 2021 +0530\n\n    {net, kernel}: Guard proc_dointvec_minmax_bpf_restricted and nuke void *priv from cpuset_fork\n\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit d4885fee112b361a3844294fe32726de718f7207\nAuthor: Maitreya29 \u003cMaitreyapatni30@gmail.com\u003e\nDate:   Sun Aug 29 21:53:14 2021 +0530\n\n    Revert \"net/compat: Add missing sock updates for SCM_RIGHTS\"\n\n    This reverts commit 34c2166235171162c55ccdc2f3f77b377da76d7c.\n\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 3b09a64b0598f3b6d296db7dfff3d8a0426afe2b\nAuthor: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\nDate:   Tue Dec 11 12:14:12 2018 +0100\n\n    bpf: fix bpf_jit_limit knob for PAGE_SIZE \u003e\u003d 64K\n\n    [ Upstream commit fdadd04931c2d7cd294dc5b2b342863f94be53a3 ]\n\n    Michael and Sandipan report:\n\n      Commit ede95a63b5 introduced a bpf_jit_limit tuneable to limit BPF\n      JIT allocations. At compile time it defaults to PAGE_SIZE * 40000,\n      and is adjusted again at init time if MODULES_VADDR is defined.\n\n      For ppc64 kernels, MODULES_VADDR isn\u0027t defined, so we\u0027re stuck with\n      the compile-time default at boot-time, which is 0x9c400000 when\n      using 64K page size. This overflows the signed 32-bit bpf_jit_limit\n      value:\n\n      root@ubuntu:/tmp# cat /proc/sys/net/core/bpf_jit_limit\n      -1673527296\n\n      and can cause various unexpected failures throughout the network\n      stack. In one case `strace dhclient eth0` reported:\n\n      setsockopt(5, SOL_SOCKET, SO_ATTACH_FILTER, {len\u003d11, filter\u003d0x105dd27f8},\n                 16) \u003d -1 ENOTSUPP (Unknown error 524)\n\n      and similar failures can be seen with tools like tcpdump. This doesn\u0027t\n      always reproduce however, and I\u0027m not sure why. The more consistent\n      failure I\u0027ve seen is an Ubuntu 18.04 KVM guest booted on a POWER9\n      host would time out on systemd/netplan configuring a virtio-net NIC\n      with no noticeable errors in the logs.\n\n    Given this and also given that in near future some architectures like\n    arm64 will have a custom area for BPF JIT image allocations we should\n    get rid of the BPF_JIT_LIMIT_DEFAULT fallback / default entirely. For\n    4.21, we have an overridable bpf_jit_alloc_exec(), bpf_jit_free_exec()\n    so therefore add another overridable bpf_jit_alloc_exec_limit() helper\n    function which returns the possible size of the memory area for deriving\n    the default heuristic in bpf_jit_charge_init().\n\n    Like bpf_jit_alloc_exec() and bpf_jit_free_exec(), the new\n    bpf_jit_alloc_exec_limit() assumes that module_alloc() is the default\n    JIT memory provider, and therefore in case archs implement their custom\n    module_alloc() we use MODULES_{END,_VADDR} for limits and otherwise for\n    vmalloc_exec() cases like on ppc64 we use VMALLOC_{END,_START}.\n\n    Additionally, for archs supporting large page sizes, we should change\n    the sysctl to be handled as long to not run into sysctl restrictions\n    in future.\n\n    Fixes: ede95a63b5e8 (\"bpf: add bpf_jit_limit knob to restrict unpriv allocations\")\n    Reported-by: Sandipan Das \u003csandipan@linux.ibm.com\u003e\n    Reported-by: Michael Roth \u003cmdroth@linux.vnet.ibm.com\u003e\n    Signed-off-by: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Tested-by: Michael Roth \u003cmdroth@linux.vnet.ibm.com\u003e\n    Signed-off-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Signed-off-by: Sasha Levin \u003csashal@kernel.org\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit b69dab2bd0c9cf508c7cf64a1de10773292fce61\nAuthor: Anay Wadhera \u003canay1018@gmail.com\u003e\nDate:   Sun May 23 18:55:08 2021 +0000\n\n    Revert \"cgroup: Disable IRQs while holding css_set_lock\"\n\n    This reverts commit ac7b270e91c7b0d1b1c5544532852b55177004f1.\n\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 94d828af144997756e881c0d931adc80511ccf40\nAuthor: Colin Cross \u003cccross@android.com\u003e\nDate:   Tue Jul 12 19:53:24 2011 -0700\n\n    cgroup: Add generic cgroup subsystem permission checks\n\n    Rather than using explicit euid \u003d\u003d 0 checks when trying to move\n    tasks into a cgroup via CFS, move permission checks into each\n    specific cgroup subsystem. If a subsystem does not specify a\n    \u0027allow_attach\u0027 handler, then we fall back to doing our checks\n    the old way.\n\n    Use the \u0027allow_attach\u0027 handler for the \u0027cpu\u0027 cgroup to allow\n    non-root processes to add arbitrary processes to a \u0027cpu\u0027 cgroup\n    if it has the CAP_SYS_NICE capability set.\n\n    This version of the patch adds a \u0027allow_attach\u0027 handler instead\n    of reusing the \u0027can_attach\u0027 handler.  If the \u0027can_attach\u0027 handler\n    is reused, a new cgroup that implements \u0027can_attach\u0027 but not\n    the permission checks could end up with no permission checks\n    at all.\n\n    Change-Id: Icfa950aa9321d1ceba362061d32dc7dfa2c64f0c\n    Original-Author: San Mehat \u003csan@google.com\u003e\n    Signed-off-by: Colin Cross \u003cccross@android.com\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 0788f4e1ff2c1d6ef4cd5fe01ae729124fe59337\nAuthor: Rom Lemarchand \u003cromlem@android.com\u003e\nDate:   Fri Nov 7 12:48:17 2014 -0800\n\n    cgroup: refactor allow_attach function into common code\n\n    move cpu_cgroup_allow_attach to a common subsys_cgroup_allow_attach.\n    This allows any process with CAP_SYS_NICE to move tasks across cgroups if\n    they use this function as their allow_attach handler.\n\n    Bug: 18260435\n    Change-Id: I6bb4933d07e889d0dc39e33b4e71320c34a2c90f\n    Signed-off-by: Rom Lemarchand \u003cromlem@android.com\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 79ebef4904c307565e9a19e56c4641e6bc3c0f23\nAuthor: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\nDate:   Fri Jan 22 16:00:56 2021 +0100\n\n    bpf: Fix buggy rsh min/max bounds tracking\n\n    [ no upstream commit ]\n\n    Fix incorrect bounds tracking for RSH opcode. Commit f23cc643f9ba (\"bpf: fix\n    range arithmetic for bpf map access\") had a wrong assumption about min/max\n    bounds. The new dst_reg-\u003emin_value needs to be derived by right shifting the\n    max_val bounds, not min_val, and likewise new dst_reg-\u003emax_value needs to be\n    derived by right shifting the min_val bounds, not max_val. Later stable kernels\n    than 4.9 are not affected since bounds tracking was overall reworked and they\n    already track this similarly as in the fix.\n\n    Fixes: f23cc643f9ba (\"bpf: fix range arithmetic for bpf map access\")\n    Reported-by: Ryota Shiga (Flatt Security)\n    Signed-off-by: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Reviewed-by: John Fastabend \u003cjohn.fastabend@gmail.com\u003e\n    Cc: Josef Bacik \u003cjbacik@fb.com\u003e\n    Signed-off-by: Greg Kroah-Hartman \u003cgregkh@linuxfoundation.org\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit e0985270080a93fc516f8ac07b7fd5c9b073e032\nAuthor: Tejun Heo \u003ctj@kernel.org\u003e\nDate:   Wed Aug 10 11:23:44 2016 -0400\n\n    cgroup: add tracepoints for basic operations\n\n    Debugging what goes wrong with cgroup setup can get hairy.  Add\n    tracepoints for cgroup hierarchy mount, cgroup creation/destruction\n    and task migration operations for better visibility.\n\n    Signed-off-by: Tejun Heo \u003ctj@kernel.org\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 333ba81a68d5c1a99cd9db9ca39fd744343a0d30\nAuthor: Daniel Bristot de Oliveira \u003cbristot@redhat.com\u003e\nDate:   Wed Jun 22 17:28:41 2016 -0300\n\n    cgroup: Disable IRQs while holding css_set_lock\n\n    While testing the deadline scheduler + cgroup setup I hit this\n    warning.\n\n    [  132.612935] ------------[ cut here ]------------\n    [  132.612951] WARNING: CPU: 5 PID: 0 at kernel/softirq.c:150 __local_bh_enable_ip+0x6b/0x80\n    [  132.612952] Modules linked in: (a ton of modules...)\n    [  132.612981] CPU: 5 PID: 0 Comm: swapper/5 Not tainted 4.7.0-rc2 #2\n    [  132.612981] Hardware name: QEMU Standard PC (i440FX + PIIX, 1996), BIOS 1.8.2-20150714_191134- 04/01/2014\n    [  132.612982]  0000000000000086 45c8bb5effdd088b ffff88013fd43da0 ffffffff813d229e\n    [  132.612984]  0000000000000000 0000000000000000 ffff88013fd43de0 ffffffff810a652b\n    [  132.612985]  00000096811387b5 0000000000000200 ffff8800bab29d80 ffff880034c54c00\n    [  132.612986] Call Trace:\n    [  132.612987]  \u003cIRQ\u003e  [\u003cffffffff813d229e\u003e] dump_stack+0x63/0x85\n    [  132.612994]  [\u003cffffffff810a652b\u003e] __warn+0xcb/0xf0\n    [  132.612997]  [\u003cffffffff810e76a0\u003e] ? push_dl_task.part.32+0x170/0x170\n    [  132.612999]  [\u003cffffffff810a665d\u003e] warn_slowpath_null+0x1d/0x20\n    [  132.613000]  [\u003cffffffff810aba5b\u003e] __local_bh_enable_ip+0x6b/0x80\n    [  132.613008]  [\u003cffffffff817d6c8a\u003e] _raw_write_unlock_bh+0x1a/0x20\n    [  132.613010]  [\u003cffffffff817d6c9e\u003e] _raw_spin_unlock_bh+0xe/0x10\n    [  132.613015]  [\u003cffffffff811388ac\u003e] put_css_set+0x5c/0x60\n    [  132.613016]  [\u003cffffffff8113dc7f\u003e] cgroup_free+0x7f/0xa0\n    [  132.613017]  [\u003cffffffff810a3912\u003e] __put_task_struct+0x42/0x140\n    [  132.613018]  [\u003cffffffff810e776a\u003e] dl_task_timer+0xca/0x250\n    [  132.613027]  [\u003cffffffff810e76a0\u003e] ? push_dl_task.part.32+0x170/0x170\n    [  132.613030]  [\u003cffffffff8111371e\u003e] __hrtimer_run_queues+0xee/0x270\n    [  132.613031]  [\u003cffffffff81113ec8\u003e] hrtimer_interrupt+0xa8/0x190\n    [  132.613034]  [\u003cffffffff81051a58\u003e] local_apic_timer_interrupt+0x38/0x60\n    [  132.613035]  [\u003cffffffff817d9b0d\u003e] smp_apic_timer_interrupt+0x3d/0x50\n    [  132.613037]  [\u003cffffffff817d7c5c\u003e] apic_timer_interrupt+0x8c/0xa0\n    [  132.613038]  \u003cEOI\u003e  [\u003cffffffff81063466\u003e] ? native_safe_halt+0x6/0x10\n    [  132.613043]  [\u003cffffffff81037a4e\u003e] default_idle+0x1e/0xd0\n    [  132.613044]  [\u003cffffffff810381cf\u003e] arch_cpu_idle+0xf/0x20\n    [  132.613046]  [\u003cffffffff810e8fda\u003e] default_idle_call+0x2a/0x40\n    [  132.613047]  [\u003cffffffff810e92d7\u003e] cpu_startup_entry+0x2e7/0x340\n    [  132.613048]  [\u003cffffffff81050235\u003e] start_secondary+0x155/0x190\n    [  132.613049] ---[ end trace f91934d162ce9977 ]---\n\n    The warn is the spin_(lock|unlock)_bh(\u0026css_set_lock) in the interrupt\n    context. Converting the spin_lock_bh to spin_lock_irq(save) to avoid\n    this problem - and other problems of sharing a spinlock with an\n    interrupt.\n\n    Cc: Tejun Heo \u003ctj@kernel.org\u003e\n    Cc: Li Zefan \u003clizefan@huawei.com\u003e\n    Cc: Johannes Weiner \u003channes@cmpxchg.org\u003e\n    Cc: Juri Lelli \u003cjuri.lelli@arm.com\u003e\n    Cc: Steven Rostedt \u003crostedt@goodmis.org\u003e\n    Cc: cgroups@vger.kernel.org\n    Cc: stable@vger.kernel.org # 4.5+\n    Cc: linux-kernel@vger.kernel.org\n    Reviewed-by: Rik van Riel \u003criel@redhat.com\u003e\n    Reviewed-by: \"Luis Claudio R. Goncalves\" \u003clgoncalv@redhat.com\u003e\n    Signed-off-by: Daniel Bristot de Oliveira \u003cbristot@redhat.com\u003e\n    Acked-by: Zefan Li \u003clizefan@huawei.com\u003e\n    Signed-off-by: Tejun Heo \u003ctj@kernel.org\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 3417b34c0a062b02b44c663c4dedad65868c2d09\nAuthor: Johannes Weiner \u003channes@cmpxchg.org\u003e\nDate:   Thu Dec 6 09:06:47 2018 -0500\n\n    FROMLIST: kernel: cgroup: add poll file operation\n\n    Cgroup has a standardized poll/notification mechanism for waking all\n    pollers on all fds when a filesystem node changes.  To allow polling for\n    custom events, add a .poll callback that can override the default.\n\n    This is in preparation for pollable cgroup pressure files which have\n    per-fd trigger configurations.\n\n    Link: http://lkml.kernel.org/r/20190124211518.244221-3-surenb@google.com\n    Signed-off-by: Johannes Weiner \u003channes@cmpxchg.org\u003e\n    Signed-off-by: Suren Baghdasaryan \u003csurenb@google.com\u003e\n    Cc: Dennis Zhou \u003cdennis@kernel.org\u003e\n    Cc: Ingo Molnar \u003cmingo@redhat.com\u003e\n    Cc: Jens Axboe \u003caxboe@kernel.dk\u003e\n    Cc: Li Zefan \u003clizefan@huawei.com\u003e\n    Cc: Peter Zijlstra \u003cpeterz@infradead.org\u003e\n    Cc: Tejun Heo \u003ctj@kernel.org\u003e\n    Signed-off-by: Andrew Morton \u003cakpm@linux-foundation.org\u003e\n    Signed-off-by: Stephen Rothwell \u003csfr@canb.auug.org.au\u003e\n\n    (in linux-next: https://git.kernel.org/pub/scm/linux/kernel/git/next/linux-next.git/commit/?id\u003dc88177361203be291a49956b6c9d5ec164ea24b2)\n\n    Conflicts:\n            include/linux/cgroup-defs.h\n            kernel/cgroup.c\n\n    1. made changes in kernel/cgroup.c instead of kernel/cgroup/cgroup.c\n    2. replaced __poll_t with unsigned int\n\n    Bug: 111308141\n    Test: modified lmkd to use PSI and tested using lmkd_unit_test\n\n    Change-Id: Ie3d914197d1f150e1d83c6206865566a7cbff1b4\n    Signed-off-by: Suren Baghdasaryan \u003csurenb@google.com\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 49d2545c922a6a2b61baddde5a85899fa357ba6a\nAuthor: Tejun Heo \u003ctj@kernel.org\u003e\nDate:   Tue Dec 27 14:49:03 2016 -0500\n\n    UPSTREAM: cgroup add cftype-\u003eopen/release() callbacks\n\n    Pipe the newly added kernfs-\u003eopen/release() callbacks through cftype.\n    While at it, as cleanup operations now can be performed from\n    -\u003erelease() instead of -\u003eseq_stop(), make the latter optional.\n\n    Signed-off-by: Tejun Heo \u003ctj@kernel.org\u003e\n    Acked-by: Acked-by: Zefan Li \u003clizefan@huawei.com\u003e\n\n    (cherry picked from commit e90cbebc3fa5caea4c8bfeb0d0157a0cee53efc7)\n\n    Bug: 111308141\n    Test: modified lmkd to use PSI and tested using lmkd_unit_test\n\n    Change-Id: Iff9794cbbc2c7067c24cb2f767bbdeffa26b5180\n    Signed-off-by: Suren Baghdasaryan \u003csurenb@google.com\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 65ef0879b938573a2f2bf1a8c26576847ffb29cd\nAuthor: Zefan Li \u003clizefan@huawei.com\u003e\nDate:   Sat May 9 11:32:10 2020 +0800\n\n    netprio_cgroup: Fix unlimited memory leak of v2 cgroups\n\n    [ Upstream commit 090e28b229af92dc5b40786ca673999d59e73056 ]\n\n    If systemd is configured to use hybrid mode which enables the use of\n    both cgroup v1 and v2, systemd will create new cgroup on both the default\n    root (v2) and netprio_cgroup hierarchy (v1) for a new session and attach\n    task to the two cgroups. If the task does some network thing then the v2\n    cgroup can never be freed after the session exited.\n\n    One of our machines ran into OOM due to this memory leak.\n\n    In the scenario described above when sk_alloc() is called\n    cgroup_sk_alloc() thought it\u0027s in v2 mode, so it stores\n    the cgroup pointer in sk-\u003esk_cgrp_data and increments\n    the cgroup refcnt, but then sock_update_netprioidx()\n    thought it\u0027s in v1 mode, so it stores netprioidx value\n    in sk-\u003esk_cgrp_data, so the cgroup refcnt will never be freed.\n\n    Currently we do the mode switch when someone writes to the ifpriomap\n    cgroup control file. The easiest fix is to also do the switch when\n    a task is attached to a new cgroup.\n\n    Fixes: bd1060a1d671 (\"sock, cgroup: add sock-\u003esk_cgroup\")\n    Reported-by: Yang Yingliang \u003cyangyingliang@huawei.com\u003e\n    Tested-by: Yang Yingliang \u003cyangyingliang@huawei.com\u003e\n    Signed-off-by: Zefan Li \u003clizefan@huawei.com\u003e\n    Acked-by: Tejun Heo \u003ctj@kernel.org\u003e\n    Signed-off-by: Jakub Kicinski \u003ckuba@kernel.org\u003e\n    Signed-off-by: Greg Kroah-Hartman \u003cgregkh@linuxfoundation.org\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 3500f4cf443c8d6aa39519c1191661aa8ebeecbc\nAuthor: Shakeel Butt \u003cshakeelb@google.com\u003e\nDate:   Mon Mar 9 22:16:05 2020 -0700\n\n    cgroup: memcg: net: do not associate sock with unrelated cgroup\n\n    [ Upstream commit e876ecc67db80dfdb8e237f71e5b43bb88ae549c ]\n\n    We are testing network memory accounting in our setup and noticed\n    inconsistent network memory usage and often unrelated cgroups network\n    usage correlates with testing workload. On further inspection, it\n    seems like mem_cgroup_sk_alloc() and cgroup_sk_alloc() are broken in\n    irq context specially for cgroup v1.\n\n    mem_cgroup_sk_alloc() and cgroup_sk_alloc() can be called in irq context\n    and kind of assumes that this can only happen from sk_clone_lock()\n    and the source sock object has already associated cgroup. However in\n    cgroup v1, where network memory accounting is opt-in, the source sock\n    can be unassociated with any cgroup and the new cloned sock can get\n    associated with unrelated interrupted cgroup.\n\n    Cgroup v2 can also suffer if the source sock object was created by\n    process in the root cgroup or if sk_alloc() is called in irq context.\n    The fix is to just do nothing in interrupt.\n\n    WARNING: Please note that about half of the TCP sockets are allocated\n    from the IRQ context, so, memory used by such sockets will not be\n    accouted by the memcg.\n\n    The stack trace of mem_cgroup_sk_alloc() from IRQ-context:\n\n    CPU: 70 PID: 12720 Comm: ssh Tainted:  5.6.0-smp-DEV #1\n    Hardware name: ...\n    Call Trace:\n     \u003cIRQ\u003e\n     dump_stack+0x57/0x75\n     mem_cgroup_sk_alloc+0xe9/0xf0\n     sk_clone_lock+0x2a7/0x420\n     inet_csk_clone_lock+0x1b/0x110\n     tcp_create_openreq_child+0x23/0x3b0\n     tcp_v6_syn_recv_sock+0x88/0x730\n     tcp_check_req+0x429/0x560\n     tcp_v6_rcv+0x72d/0xa40\n     ip6_protocol_deliver_rcu+0xc9/0x400\n     ip6_input+0x44/0xd0\n     ? ip6_protocol_deliver_rcu+0x400/0x400\n     ip6_rcv_finish+0x71/0x80\n     ipv6_rcv+0x5b/0xe0\n     ? ip6_sublist_rcv+0x2e0/0x2e0\n     process_backlog+0x108/0x1e0\n     net_rx_action+0x26b/0x460\n     __do_softirq+0x104/0x2a6\n     do_softirq_own_stack+0x2a/0x40\n     \u003c/IRQ\u003e\n     do_softirq.part.19+0x40/0x50\n     __local_bh_enable_ip+0x51/0x60\n     ip6_finish_output2+0x23d/0x520\n     ? ip6table_mangle_hook+0x55/0x160\n     __ip6_finish_output+0xa1/0x100\n     ip6_finish_output+0x30/0xd0\n     ip6_output+0x73/0x120\n     ? __ip6_finish_output+0x100/0x100\n     ip6_xmit+0x2e3/0x600\n     ? ipv6_anycast_cleanup+0x50/0x50\n     ? inet6_csk_route_socket+0x136/0x1e0\n     ? skb_free_head+0x1e/0x30\n     inet6_csk_xmit+0x95/0xf0\n     __tcp_transmit_skb+0x5b4/0xb20\n     __tcp_send_ack.part.60+0xa3/0x110\n     tcp_send_ack+0x1d/0x20\n     tcp_rcv_state_process+0xe64/0xe80\n     ? tcp_v6_connect+0x5d1/0x5f0\n     tcp_v6_do_rcv+0x1b1/0x3f0\n     ? tcp_v6_do_rcv+0x1b1/0x3f0\n     __release_sock+0x7f/0xd0\n     release_sock+0x30/0xa0\n     __inet_stream_connect+0x1c3/0x3b0\n     ? prepare_to_wait+0xb0/0xb0\n     inet_stream_connect+0x3b/0x60\n     __sys_connect+0x101/0x120\n     ? __sys_getsockopt+0x11b/0x140\n     __x64_sys_connect+0x1a/0x20\n     do_syscall_64+0x51/0x200\n     entry_SYSCALL_64_after_hwframe+0x44/0xa9\n\n    The stack trace of mem_cgroup_sk_alloc() from IRQ-context:\n    Fixes: 2d7580738345 (\"mm: memcontrol: consolidate cgroup socket tracking\")\n    Fixes: d979a39d7242 (\"cgroup: duplicate cgroup reference when cloning sockets\")\n    Signed-off-by: Shakeel Butt \u003cshakeelb@google.com\u003e\n    Reviewed-by: Roman Gushchin \u003cguro@fb.com\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Greg Kroah-Hartman \u003cgregkh@linuxfoundation.org\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 9e585103bce276b4db0b6a130d72dcacd1d9ab20\nAuthor: Yang Yingliang \u003cyangyingliang@huawei.com\u003e\nDate:   Thu Aug 13 20:27:57 2020 +0000\n\n    cgroup: add missing skcd-\u003eno_refcnt check in cgroup_sk_clone()\n\n    Add skcd-\u003eno_refcnt check which is missed when backporting\n    ad0f75e5f57c (\"cgroup: fix cgroup_sk_alloc() for sk_clone_lock()\").\n\n    This patch is needed in stable-4.9, stable-4.14 and stable-4.19.\n\n    Signed-off-by: Yang Yingliang \u003cyangyingliang@huawei.com\u003e\n    Signed-off-by: Sasha Levin \u003csashal@kernel.org\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 2c5b51f1bb6b73cf14ece028a035329e6a1d748e\nAuthor: Cong Wang \u003cxiyou.wangcong@gmail.com\u003e\nDate:   Thu Jul 2 11:52:56 2020 -0700\n\n    cgroup: fix cgroup_sk_alloc() for sk_clone_lock()\n\n    [ Upstream commit ad0f75e5f57ccbceec13274e1e242f2b5a6397ed ]\n\n    When we clone a socket in sk_clone_lock(), its sk_cgrp_data is\n    copied, so the cgroup refcnt must be taken too. And, unlike the\n    sk_alloc() path, sock_update_netprioidx() is not called here.\n    Therefore, it is safe and necessary to grab the cgroup refcnt\n    even when cgroup_sk_alloc is disabled.\n\n    sk_clone_lock() is in BH context anyway, the in_interrupt()\n    would terminate this function if called there. And for sk_alloc()\n    skcd-\u003eval is always zero. So it\u0027s safe to factor out the code\n    to make it more readable.\n\n    The global variable \u0027cgroup_sk_alloc_disabled\u0027 is used to determine\n    whether to take these reference counts. It is impossible to make\n    the reference counting correct unless we save this bit of information\n    in skcd-\u003eval. So, add a new bit there to record whether the socket\n    has already taken the reference counts. This obviously relies on\n    kmalloc() to align cgroup pointers to at least 4 bytes,\n    ARCH_KMALLOC_MINALIGN is certainly larger than that.\n\n    This bug seems to be introduced since the beginning, commit\n    d979a39d7242 (\"cgroup: duplicate cgroup reference when cloning sockets\")\n    tried to fix it but not compeletely. It seems not easy to trigger until\n    the recent commit 090e28b229af\n    (\"netprio_cgroup: Fix unlimited memory leak of v2 cgroups\") was merged.\n\n    Fixes: bd1060a1d671 (\"sock, cgroup: add sock-\u003esk_cgroup\")\n    Reported-by: Cameron Berkenpas \u003ccam@neo-zeon.de\u003e\n    Reported-by: Peter Geis \u003cpgwipeout@gmail.com\u003e\n    Reported-by: Lu Fengqi \u003clufq.fnst@cn.fujitsu.com\u003e\n    Reported-by: Daniël Sonck \u003cdsonck92@gmail.com\u003e\n    Reported-by: Zhang Qiang \u003cqiang.zhang@windriver.com\u003e\n    Tested-by: Cameron Berkenpas \u003ccam@neo-zeon.de\u003e\n    Tested-by: Peter Geis \u003cpgwipeout@gmail.com\u003e\n    Tested-by: Thomas Lamprecht \u003ct.lamprecht@proxmox.com\u003e\n    Cc: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Cc: Zefan Li \u003clizefan@huawei.com\u003e\n    Cc: Tejun Heo \u003ctj@kernel.org\u003e\n    Cc: Roman Gushchin \u003cguro@fb.com\u003e\n    Signed-off-by: Cong Wang \u003cxiyou.wangcong@gmail.com\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Greg Kroah-Hartman \u003cgregkh@linuxfoundation.org\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 0769a838c5c8049ec876f703f2f5812ae1881066\nAuthor: Chenbo Feng \u003cfengc@google.com\u003e\nDate:   Wed Mar 22 17:27:35 2017 -0700\n\n    BACKPORT: UPSTREAM: Add a eBPF helper function to retrieve socket uid\n\n    Cherry-pick from commit 6acc5c2910689fc6ee181bf63085c5efff6a42bd\n\n    Returns the owner uid of the socket inside a sk_buff. This is useful to\n    perform per-UID accounting of network traffic or per-UID packet\n    filtering. The socket need to be a fullsock otherwise overflowuid is\n    returned.\n\n    Signed-off-by: Chenbo Feng \u003cfengc@google.com\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Bug: 30950746\n    Change-Id: Idc00947ccfdd4e9f2214ffc4178d701cd9ead0ac\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit b042c02048e1a9d9860c409ed638d61e6f36cb9c\nAuthor: Chenbo Feng \u003cfengc@google.com\u003e\nDate:   Wed Mar 22 17:27:34 2017 -0700\n\n    BACKPORT: UPSTREAM: Add a helper function to get socket cookie in eBPF\n\n    Cherrypick from commit: 91b8270f2a4d1d9b268de90451cdca63a70052d6\n\n    Retrieve the socket cookie generated by sock_gen_cookie() from a sk_buff\n    with a known socket. Generates a new cookie if one was not yet set.If\n    the socket pointer inside sk_buff is NULL, 0 is returned. The helper\n    function coud be useful in monitoring per socket networking traffic\n    statistics and provide a unique socket identifier per namespace.\n\n    Acked-by: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Acked-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Acked-by: Willem de Bruijn \u003cwillemb@google.com\u003e\n    Signed-off-by: Chenbo Feng \u003cfengc@google.com\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Bug: 30950746\n    Change-Id: I95918dcc3ceffb3061495a859d28aee88e3cde3c\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit d37171aecd9d5fc669d3811a0283d594957bda2d\nAuthor: Chenbo Feng \u003cfengc@google.com\u003e\nDate:   Wed May 3 15:22:42 2017 -0700\n\n    ANDROID: Fix missing uapi headers\n\n    Update the missing bpf helper function name in bpf_func_id to keep the\n    uapi header consistent with upstream uapi header because we need the\n    new added bpf helper function bpf get_socket_cookie and get_socket_uid.\n    The patch related to those headers are not backetported since they are\n    not related and backport them will bring in extra confilict.\n\n    Signed-off-by: Chenbo Feng \u003cfengc@google.com\u003e\n    Bug: 30950746\n    Change-Id: I2b5fd03799ac5f2e3243ab11a1bccb932f06c312\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 2b5c5bedb9701ba1fcb8e54067a856aeb103fb98\nAuthor: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\nDate:   Fri Sep 23 01:28:37 2016 +0200\n\n    bpf: add helper to invalidate hash\n\n    Add a small helper that complements 36bbef52c7eb (\"bpf: direct packet\n    write and access for helpers for clsact progs\") for invalidating the\n    current skb-\u003ehash after mangling on headers via direct packet write.\n\n    Signed-off-by: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Acked-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 8f8fdc7db3b78616b38a12c53e2ae2944fb944f4\nAuthor: Anay Wadhera \u003cawadhera@berkeley.edu\u003e\nDate:   Fri May 21 15:18:35 2021 -0700\n\n    net: take compile fix from 4.9\n\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 8138272a39a5a3ee29a65b1816f3216d882f9f95\nAuthor: Anay Wadhera \u003cawadhera@berkeley.edu\u003e\nDate:   Fri May 21 15:06:45 2021 -0700\n\n    cgroup: replace out_idr_free with actual code\n\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 2aad4250db090889c674ffd3b3ec1583158a0c96\nAuthor: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\nDate:   Wed Mar 9 03:00:02 2016 +0100\n\n    ip_tunnel: add support for setting flow label via collect metadata\n\n    This patch extends udp_tunnel6_xmit_skb() to pass in the IPv6 flow label\n    from call sites. Currently, there\u0027s no such option and it\u0027s always set to\n    zero when writing ip6_flow_hdr(). Add a label member to ip_tunnel_key, so\n    that flow-based tunnels via collect metadata frontends can make use of it.\n    vxlan and geneve will be converted to add flow label support separately.\n\n    Signed-off-by: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 34c6c1f1c1a32901c4db911ae2d29ed9778b4b1b\nAuthor: Jamal Hadi Salim \u003cjhs@mojatatu.com\u003e\nDate:   Sat Jul 2 06:43:14 2016 -0400\n\n    net: simplify and make pkt_type_ok() available for other users\n\n    Suggested-by: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Signed-off-by: Jamal Hadi Salim \u003cjhs@mojatatu.com\u003e\n    Acked-by: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit f37b41508db5d2dc8ea95938d337a6cd109fddec\nAuthor: Aditya Kali \u003cadityakali@google.com\u003e\nDate:   Fri Jan 29 02:54:08 2016 -0600\n\n    kernfs: define kernfs_node_dentry\n\n    Add a new kernfs api is added to lookup the dentry for a particular\n    kernfs path.\n\n    Signed-off-by: Aditya Kali \u003cadityakali@google.com\u003e\n    Signed-off-by: Serge E. Hallyn \u003cserge.hallyn@canonical.com\u003e\n    Acked-by: Greg Kroah-Hartman \u003cgregkh@linuxfoundation.org\u003e\n    Signed-off-by: Tejun Heo \u003ctj@kernel.org\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 868300a337fd8c73f5b6a87192bf7b8a0cdae675\nAuthor: Tejun Heo \u003ctj@kernel.org\u003e\nDate:   Fri Aug 11 05:49:01 2017 -0700\n\n    BACKPORT: cgroup: misc changes\n\n    Misc trivial changes to prepare for future changes.  No functional\n    difference.\n\n    * Expose cgroup_get(), cgroup_tryget() and cgroup_parent().\n\n    * Implement task_dfl_cgroup() which dereferences css_set-\u003edfl_cgrp.\n\n    * Rename cgroup_stats_show() to cgroup_stat_show() for consistency\n      with the file name.\n\n    Signed-off-by: Tejun Heo \u003ctj@kernel.org\u003e\n\n    (cherry picked from commit 3e48930cc74f0c212ee1838f89ad0ca7fcf2fea1)\n\n    Conflicts:\n            kernel/cgroup/cgroup.c\n\n    (1. manual merge because kernel/cgroup/cgroup.c is under kernel/cgroup.c\n    2. cgroup_stats_show change is skipped because the function dos not exist)\n\n    Bug: 111308141\n    Test: modified lmkd to use PSI and tested using lmkd_unit_test\n\n    Signed-off-by: Suren Baghdasaryan \u003csurenb@google.com\u003e\n    Change-Id: I756ee3dcf0d0f3da69cd1b58e644271625053538\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit ad8b6378eb7539914b444aa63a1d8f8665e1749e\nAuthor: Josh Poimboeuf \u003cjpoimboe@redhat.com\u003e\nDate:   Wed Mar 1 12:04:44 2017 -0600\n\n    objtool, modules: Discard objtool annotation sections for modules\n\n    commit e390f9a9689a42f477a6073e2e7df530a4c1b740 upstream.\n\n    The \u0027__unreachable\u0027 and \u0027__func_stack_frame_non_standard\u0027 sections are\n    only used at compile time.  They\u0027re discarded for vmlinux but they\n    should also be discarded for modules.\n\n    Since this is a recurring pattern, prefix the section names with\n    \".discard.\".  It\u0027s a nice convention and vmlinux.lds.h already discards\n    such sections.\n\n    Also remove the \u0027a\u0027 (allocatable) flag from the __unreachable section\n    since it doesn\u0027t make sense for a discarded section.\n\n    Suggested-by: Linus Torvalds \u003ctorvalds@linux-foundation.org\u003e\n    Signed-off-by: Josh Poimboeuf \u003cjpoimboe@redhat.com\u003e\n    Cc: Jessica Yu \u003cjeyu@redhat.com\u003e\n    Cc: Peter Zijlstra \u003cpeterz@infradead.org\u003e\n    Cc: Thomas Gleixner \u003ctglx@linutronix.de\u003e\n    Fixes: d1091c7fa3d5 (\"objtool: Improve detection of BUG() and other dead ends\")\n    Link: http://lkml.kernel.org/r/20170301180444.lhd53c5tibc4ns77@treble\n    Signed-off-by: Ingo Molnar \u003cmingo@kernel.org\u003e\n    [dwmw2: Remove the unreachable part in backporting since it\u0027s not here yet]\n    Signed-off-by: David Woodhouse \u003cdwmw@amazon.co.ku\u003e\n    Signed-off-by: Greg Kroah-Hartman \u003cgregkh@linuxfoundation.org\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 7875074852deaeb63000514a606b385bccac5ffd\nAuthor: Josh Poimboeuf \u003cjpoimboe@redhat.com\u003e\nDate:   Sun Feb 28 22:22:35 2016 -0600\n\n    objtool: Add STACK_FRAME_NON_STANDARD() macro\n\n    Add a new macro, STACK_FRAME_NON_STANDARD(), which is used to denote a\n    function which does something unusual related to its stack frame.  Use\n    of the macro prevents objtool from emitting a false positive warning.\n\n    Signed-off-by: Josh Poimboeuf \u003cjpoimboe@redhat.com\u003e\n    Cc: Andrew Morton \u003cakpm@linux-foundation.org\u003e\n    Cc: Andy Lutomirski \u003cluto@kernel.org\u003e\n    Cc: Arnaldo Carvalho de Melo \u003cacme@kernel.org\u003e\n    Cc: Bernd Petrovitsch \u003cbernd@petrovitsch.priv.at\u003e\n    Cc: Borislav Petkov \u003cbp@alien8.de\u003e\n    Cc: Chris J Arges \u003cchris.j.arges@canonical.com\u003e\n    Cc: Jiri Slaby \u003cjslaby@suse.cz\u003e\n    Cc: Linus Torvalds \u003ctorvalds@linux-foundation.org\u003e\n    Cc: Michal Marek \u003cmmarek@suse.cz\u003e\n    Cc: Namhyung Kim \u003cnamhyung@gmail.com\u003e\n    Cc: Pedro Alves \u003cpalves@redhat.com\u003e\n    Cc: Peter Zijlstra \u003cpeterz@infradead.org\u003e\n    Cc: Thomas Gleixner \u003ctglx@linutronix.de\u003e\n    Cc: live-patching@vger.kernel.org\n    Link: http://lkml.kernel.org/r/34487a17b23dba43c50941599d47054a9584b219.1456719558.git.jpoimboe@redhat.com\n    Signed-off-by: Ingo Molnar \u003cmingo@kernel.org\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 2c3ed942b62c8c552b90af9ec5f2e33fe6f8e263\nAuthor: Anay Wadhera \u003cawadhera@berkeley.edu\u003e\nDate:   Fri May 21 14:45:03 2021 -0700\n\n    remove leftovers from 6ea07b4590d3174a53303\n\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 7a851110e4899b9d6e0f4b2765131f086671017d\nAuthor: Eric W. Biederman \u003cebiederm@xmission.com\u003e\nDate:   Tue May 24 09:29:01 2016 -0500\n\n    fs: Add user namespace member to struct super_block\n\n    Start marking filesystems with a user namespace owner, s_user_ns.  In\n    this change this is only used for permission checks of who may mount a\n    filesystem.  Ultimately s_user_ns will be used for translating ids and\n    checking capabilities for filesystems mounted from user namespaces.\n\n    The default policy for setting s_user_ns is implemented in sget(),\n    which arranges for s_user_ns to be set to current_user_ns() and to\n    ensure that the mounter of the filesystem has CAP_SYS_ADMIN in that\n    user_ns.\n\n    The guts of sget are split out into another function sget_userns().\n    The function sget_userns calls alloc_super with the specified user\n    namespace or it verifies the existing superblock that was found\n    has the expected user namespace, and fails with EBUSY when it is not.\n    This failing prevents users with the wrong privileges mounting a\n    filesystem.\n\n    The reason for the split of sget_userns from sget is that in some\n    cases such as mount_ns and kernfs_mount_ns a different policy for\n    permission checking of mounts and setting s_user_ns is necessary, and\n    the existence of sget_userns() allows those policies to be\n    implemented.\n\n    The helper mount_ns is expected to be used for filesystems such as\n    proc and mqueuefs which present per namespace information.  The\n    function mount_ns is modified to call sget_userns instead of sget to\n    ensure the user namespace owner of the namespace whose information is\n    presented by the filesystem is used on the superblock.\n\n    For sysfs and cgroup the appropriate permission checks are already in\n    place, and kernfs_mount_ns is modified to call sget_userns so that\n    the init_user_ns is the only user namespace used.\n\n    For the cgroup filesystem cgroup namespace mounts are bind mounts of a\n    subset of the full cgroup filesystem and as such s_user_ns must be the\n    same for all of them as there is only a single superblock.\n\n    Mounts of sysfs that vary based on the network namespace could in principle\n    change s_user_ns but it keeps the analysis and implementation of kernfs\n    simpler if that is not supported, and at present there appear to be no\n    benefits from supporting a different s_user_ns on any sysfs mount.\n\n    Getting the details of setting s_user_ns correct has been\n    a long process.  Thanks to Pavel Tikhorirorv who spotted a leak\n    in sget_userns.  Thanks to Seth Forshee who has kept the work alive.\n\n    Thanks-to: Seth Forshee \u003cseth.forshee@canonical.com\u003e\n    Thanks-to: Pavel Tikhomirov \u003cptikhomirov@virtuozzo.com\u003e\n    Acked-by: Seth Forshee \u003cseth.forshee@canonical.com\u003e\n    Signed-off-by: Eric W. Biederman \u003cebiederm@xmission.com\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 0c4c71d19f69a92aa67affe258ca3aa3741ad31e\nAuthor: Eric W. Biederman \u003cebiederm@xmission.com\u003e\nDate:   Mon May 23 14:51:59 2016 -0500\n\n    vfs: Pass data, ns, and ns-\u003euserns to mount_ns\n\n    Today what is normally called data (the mount options) is not passed\n    to fill_super through mount_ns.\n\n    Pass the mount options and the namespace separately to mount_ns so\n    that filesystems such as proc that have mount options, can use\n    mount_ns.\n\n    Pass the user namespace to mount_ns so that the standard permission\n    check that verifies the mounter has permissions over the namespace can\n    be performed in mount_ns instead of in each filesystems .mount method.\n    Thus removing the duplication between mqueuefs and proc in terms of\n    permission checks.  The extra permission check does not currently\n    affect the rpc_pipefs filesystem and the nfsd filesystem as those\n    filesystems do not currently allow unprivileged mounts.  Without\n    unpvileged mounts it is guaranteed that the caller has already passed\n    capable(CAP_SYS_ADMIN) which guarantees extra permission check will\n    pass.\n\n    Update rpc_pipefs and the nfsd filesystem to ensure that the network\n    namespace reference is always taken in fill_super and always put in kill_sb\n    so that the logic is simpler and so that errors originating inside of\n    fill_super do not cause a network namespace leak.\n\n    Acked-by: Seth Forshee \u003cseth.forshee@canonical.com\u003e\n    Signed-off-by: \"Eric W. Biederman\" \u003cebiederm@xmission.com\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 1c1dc68943c9d6d5a181081e470ba4b1022f9ef4\nAuthor: Tejun Heo \u003ctj@kernel.org\u003e\nDate:   Wed Aug 10 11:23:44 2016 -0400\n\n    kernfs: make kernfs_path*() behave in the style of strlcpy()\n\n    kernfs_path*() functions always return the length of the full path but\n    the path content is undefined if the length is larger than the\n    provided buffer.  This makes its behavior different from strlcpy() and\n    requires error handling in all its users even when they don\u0027t care\n    about truncation.  In addition, the implementation can actully be\n    simplified by making it behave properly in strlcpy() style.\n\n    * Update kernfs_path_from_node_locked() to always fill up the buffer\n      with path.  If the buffer is not large enough, the output is\n      truncated and terminated.\n\n    * kernfs_path() no longer needs error handling.  Make it a simple\n      inline wrapper around kernfs_path_from_node().\n\n    * sysfs_warn_dup()\u0027s use of kernfs_path() doesn\u0027t need error handling.\n      Updated accordingly.\n\n    * cgroup_path()\u0027s use of kernfs_path() updated to retain the old\n      behavior.\n\n    Signed-off-by: Tejun Heo \u003ctj@kernel.org\u003e\n    Acked-by: Greg Kroah-Hartman \u003cgregkh@linuxfoundation.org\u003e\n    Acked-by: Serge Hallyn \u003cserge.hallyn@ubuntu.com\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 2d96590ea44abecaabbf9f070cff5a8c8fa6d12f\nAuthor: Serge Hallyn \u003cserge.hallyn@ubuntu.com\u003e\nDate:   Sun Apr 17 15:04:31 2016 -0500\n\n    kernfs_path_from_node_locked: don\u0027t overwrite nlen\n\n    We\u0027ve calculated @len to be the bytes we need for \u0027/..\u0027 entries from\n    @kn_from to the common ancestor, and calculated @nlen to be the extra\n    bytes we need to get from the common ancestor to @kn_to.  We use them\n    as such at the end.  But in the loop copying the actual entries, we\n    overwrite @nlen.  Use a temporary variable for that instead.\n\n    Without this, the return length, when the buffer is large enough, is\n    wrong.  (When the buffer is NULL or too small, the returned value is\n    correct. The buffer contents are also correct.)\n\n    Interestingly, no callers of this function are affected by this as of\n    yet.  However the upcoming cgroup_show_path() will be.\n\n    Signed-off-by: Serge Hallyn \u003cserge.hallyn@ubuntu.com\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 9d93b38a76d5d9173ef34f36569f749cd6a2a697\nAuthor: Anay Wadhera \u003cawadhera@berkeley.edu\u003e\nDate:   Fri May 21 14:29:50 2021 -0700\n\n    arm64: bpf_jit_comp: drop artifact\n\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 594aeac0e18946504f1888fd54c68d3d8f0f8ada\nAuthor: Arnd Bergmann \u003carnd@arndb.de\u003e\nDate:   Tue Jan 10 13:08:06 2017 +0100\n\n    UPSTREAM: cgroup: move CONFIG_SOCK_CGROUP_DATA to init/Kconfig\n\n    We now \u0027select SOCK_CGROUP_DATA\u0027 but Kconfig complains that this is\n    not right when CONFIG_NET is disabled and there is no socket interface:\n\n    warning: (CGROUP_BPF) selects SOCK_CGROUP_DATA which has unmet direct dependencies (NET)\n\n    I don\u0027t know what the correct solution for this is, but simply removing\n    the dependency on NET from SOCK_CGROUP_DATA by moving it out of the\n    \u0027if NET\u0027 section avoids the warning and does not produce other build\n    errors.\n\n    Fixes: 483c4933ea09 (\"cgroup: Fix CGROUP_BPF config\")\n    Signed-off-by: Arnd Bergmann \u003carnd@arndb.de\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n\n    Fixes: Change-Id: Ib41ef78fba02eb9e592558ddbf06f9ec0aa337b6\n           (\"UPSTREAM: cgroup: Fix CGROUP_BPF config\")\n    (cherry picked from commit 73b351473547e543e9c8166dd67fd99c64c15b0b)\n    Signed-off-by: Amit Pundir \u003camit.pundir@linaro.org\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 13efe07f3d1f009cd2c2babe6169447585bd32aa\nAuthor: Andy Lutomirski \u003cluto@kernel.org\u003e\nDate:   Fri Dec 16 08:33:45 2016 -0800\n\n    UPSTREAM: cgroup: Fix CGROUP_BPF config\n\n    Cherry-pick from commit 483c4933ea09b7aa625b9d64af286fc22ec7e419\n\n    CGROUP_BPF depended on SOCK_CGROUP_DATA which can\u0027t be manually\n    enabled, making it rather challenging to turn CGROUP_BPF on.\n\n    Signed-off-by: Andy Lutomirski \u003cluto@kernel.org\u003e\n    Acked-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Bug: 30950746\n    Change-Id: Ib41ef78fba02eb9e592558ddbf06f9ec0aa337b6\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 210d70ce9fb35bdec6fb41defa232ffef6383dd4\nAuthor: Yonghong Song \u003cyhs@fb.com\u003e\nDate:   Mon Oct 23 23:53:08 2017 -0700\n\n    BACKPORT: bpf: permit multiple bpf attachments for a single perf event\n\n    This patch enables multiple bpf attachments for a\n    kprobe/uprobe/tracepoint single trace event.\n    Each trace_event keeps a list of attached perf events.\n    When an event happens, all attached bpf programs will\n    be executed based on the order of attachment.\n\n    A global bpf_event_mutex lock is introduced to protect\n    prog_array attaching and detaching. An alternative will\n    be introduce a mutex lock in every trace_event_call\n    structure, but it takes a lot of extra memory.\n    So a global bpf_event_mutex lock is a good compromise.\n\n    The bpf prog detachment involves allocation of memory.\n    If the allocation fails, a dummy do-nothing program\n    will replace to-be-detached program in-place.\n\n    Signed-off-by: Yonghong Song \u003cyhs@fb.com\u003e\n    Acked-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Acked-by: Martin KaFai Lau \u003ckafai@fb.com\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    (cherry picked from commit e87c6bc3852b981e71c757be20771546ce9f76f3)\n    Signed-off-by: Connor O\u0027Brien \u003cconnoro@google.com\u003e\n    Bug: 121213201\n    Bug: 138317270\n    Test: build \u0026 boot cuttlefish; attach 2 progs to 1 tracepoint\n    Change-Id: I390d8c0146888ddb1aed5a6f6e5dae7ef394ebc9\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 171b5880e473c4bb6bbe4ac54bd7446aa403119e\nAuthor: Alexei Starovoitov \u003cast@fb.com\u003e\nDate:   Mon Apr 18 20:11:50 2016 -0700\n\n    perf, bpf: minimize the size of perf_trace_() tracepoint handler\n\n    move trace_call_bpf() into helper function to minimize the size\n    of perf_trace_*() tracepoint handlers.\n        text\t   data\t    bss\t    dec\t \t   hex\tfilename\n    10541679\t5526646\t2945024\t19013349\t1221ee5\tvmlinux_before\n    10509422\t5526646\t2945024\t18981092\t121a0e4\tvmlinux_after\n\n    It may seem that perf_fetch_caller_regs() can also be moved,\n    but that is incorrect, since ip/sp will be wrong.\n\n    bpf+tracepoint performance is not affected, since\n    perf_swevent_put_recursion_context() is now inlined.\n    export_symbol_gpl can also be dropped.\n\n    No measurable change in normal perf tracepoints.\n\n    Suggested-by: Steven Rostedt \u003crostedt@goodmis.org\u003e\n    Signed-off-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Acked-by: Peter Zijlstra (Intel) \u003cpeterz@infradead.org\u003e\n    Acked-by: Steven Rostedt \u003crostedt@goodmis.org\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 01ff66a68246938515f9380cf0199f85ba2de95b\nAuthor: Yonghong Song \u003cyhs@fb.com\u003e\nDate:   Mon Oct 23 23:53:07 2017 -0700\n\n    UPSTREAM: bpf: use the same condition in perf event set/free bpf handler\n\n    This is a cleanup such that doing the same check in\n    perf_event_free_bpf_prog as we already do in\n    perf_event_set_bpf_prog step.\n\n    Signed-off-by: Yonghong Song \u003cyhs@fb.com\u003e\n    Acked-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Acked-by: Martin KaFai Lau \u003ckafai@fb.com\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    (cherry picked from commit 0b4c6841fee03e096b735074a0c4aab3a8e92986)\n    Signed-off-by: Connor O\u0027Brien \u003cconnoro@google.com\u003e\n    Bug: 121213201\n    Bug: 138317270\n    Test: build \u0026 boot cuttlefish\n    Change-Id: Ie423e73a73be29e8ef50cc22dbb03e14e241c8de\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 48644c6700cfe80a2773a1e89740face00ca8906\nAuthor: Alexei Starovoitov \u003cast@fb.com\u003e\nDate:   Mon Oct 2 22:50:21 2017 -0700\n\n    BACKPORT: bpf: multi program support for cgroup+bpf\n\n    introduce BPF_F_ALLOW_MULTI flag that can be used to attach multiple\n    bpf programs to a cgroup.\n\n    The difference between three possible flags for BPF_PROG_ATTACH command:\n    - NONE(default): No further bpf programs allowed in the subtree.\n    - BPF_F_ALLOW_OVERRIDE: If a sub-cgroup installs some bpf program,\n      the program in this cgroup yields to sub-cgroup program.\n    - BPF_F_ALLOW_MULTI: If a sub-cgroup installs some bpf program,\n      that cgroup program gets run in addition to the program in this cgroup.\n\n    NONE and BPF_F_ALLOW_OVERRIDE existed before. This patch doesn\u0027t\n    change their behavior. It only clarifies the semantics in relation\n    to new flag.\n\n    Only one program is allowed to be attached to a cgroup with\n    NONE or BPF_F_ALLOW_OVERRIDE flag.\n    Multiple programs are allowed to be attached to a cgroup with\n    BPF_F_ALLOW_MULTI flag. They are executed in FIFO order\n    (those that were attached first, run first)\n    The programs of sub-cgroup are executed first, then programs of\n    this cgroup and then programs of parent cgroup.\n    All eligible programs are executed regardless of return code from\n    earlier programs.\n\n    To allow efficient execution of multiple programs attached to a cgroup\n    and to avoid penalizing cgroups without any programs attached\n    introduce \u0027struct bpf_prog_array\u0027 which is RCU protected array\n    of pointers to bpf programs.\n\n    Signed-off-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Acked-by: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Acked-by: Martin KaFai Lau \u003ckafai@fb.com\u003e\n    for cgroup bits\n    Acked-by: Tejun Heo \u003ctj@kernel.org\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    (cherry picked from commit 324bda9e6c5add86ba2e1066476481c48132aca0)\n    Signed-off-by: Connor O\u0027Brien \u003cconnoro@google.com\u003e\n    Bug: 121213201\n    Bug: 138317270\n    Test: build \u0026 boot cuttlefish\n    Change-Id: I06b71c850b9f3e052b106abab7a4a3add012a3f8\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit ec820a2fad007bb6691c60787ffe1f83f31149c4\nAuthor: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\nDate:   Sat Aug 17 00:00:08 2019 +0100\n\n    bpf: add bpf_jit_limit knob to restrict unpriv allocations\n\n    commit ede95a63b5e84ddeea6b0c473b36ab8bfd8c6ce3 upstream.\n\n    Rick reported that the BPF JIT could potentially fill the entire module\n    space with BPF programs from unprivileged users which would prevent later\n    attempts to load normal kernel modules or privileged BPF programs, for\n    example. If JIT was enabled but unsuccessful to generate the image, then\n    before commit 290af86629b2 (\"bpf: introduce BPF_JIT_ALWAYS_ON config\")\n    we would always fall back to the BPF interpreter. Nowadays in the case\n    where the CONFIG_BPF_JIT_ALWAYS_ON could be set, then the load will abort\n    with a failure since the BPF interpreter was compiled out.\n\n    Add a global limit and enforce it for unprivileged users such that in case\n    of BPF interpreter compiled out we fail once the limit has been reached\n    or we fall back to BPF interpreter earlier w/o using module mem if latter\n    was compiled in. In a next step, fair share among unprivileged users can\n    be resolved in particular for the case where we would fail hard once limit\n    is reached.\n\n    Fixes: 290af86629b2 (\"bpf: introduce BPF_JIT_ALWAYS_ON config\")\n    Fixes: 0a14842f5a3c (\"net: filter: Just In Time compiler for x86-64\")\n    Co-Developed-by: Rick Edgecombe \u003crick.p.edgecombe@intel.com\u003e\n    Signed-off-by: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Acked-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Cc: Eric Dumazet \u003ceric.dumazet@gmail.com\u003e\n    Cc: Jann Horn \u003cjannh@google.com\u003e\n    Cc: Kees Cook \u003ckeescook@chromium.org\u003e\n    Cc: LKML \u003clinux-kernel@vger.kernel.org\u003e\n    Signed-off-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    [bwh: Backported to 4.9: adjust context]\n    Signed-off-by: Ben Hutchings \u003cben.hutchings@codethink.co.uk\u003e\n    Signed-off-by: Greg Kroah-Hartman \u003cgregkh@linuxfoundation.org\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 94a01e3257754b5265545453f1b27444a2ee10b3\nAuthor: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\nDate:   Fri Aug 16 23:59:56 2019 +0100\n\n    bpf: restrict access to core bpf sysctls\n\n    commit 2e4a30983b0f9b19b59e38bbf7427d7fdd480d98 upstream.\n\n    Given BPF reaches far beyond just networking these days, it was\n    never intended to allow setting and in some cases reading those\n    knobs out of a user namespace root running without CAP_SYS_ADMIN,\n    thus tighten such access.\n\n    Also the bpf_jit_enable \u003d 2 debugging mode should only be allowed\n    if kptr_restrict is not set since it otherwise can leak addresses\n    to the kernel log. Dump a note to the kernel log that this is for\n    debugging JITs only when enabled.\n\n    Signed-off-by: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Acked-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Signed-off-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    [bwh: Backported to 4.9:\n     - We don\u0027t have bpf_dump_raw_ok(), so drop the condition based on it. This\n       condition only made it a bit harder for a privileged user to do something\n       silly.\n     - Drop change to bpf_jit_kallsyms]\n    Signed-off-by: Ben Hutchings \u003cben.hutchings@codethink.co.uk\u003e\n    Signed-off-by: Greg Kroah-Hartman \u003cgregkh@linuxfoundation.org\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 1f20c7e5888e71b46d15831d7a42c36d084147a2\nAuthor: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\nDate:   Fri Aug 16 23:59:20 2019 +0100\n\n    bpf: get rid of pure_initcall dependency to enable jits\n\n    commit fa9dd599b4dae841924b022768354cfde9affecb upstream.\n\n    Having a pure_initcall() callback just to permanently enable BPF\n    JITs under CONFIG_BPF_JIT_ALWAYS_ON is unnecessary and could leave\n    a small race window in future where JIT is still disabled on boot.\n    Since we know about the setting at compilation time anyway, just\n    initialize it properly there. Also consolidate all the individual\n    bpf_jit_enable variables into a single one and move them under one\n    location. Moreover, don\u0027t allow for setting unspecified garbage\n    values on them.\n\n    Signed-off-by: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Acked-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Signed-off-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    [bwh: Backported to 4.9 as dependency of commit 2e4a30983b0f\n     \"bpf: restrict access to core bpf sysctls\":\n     - Drop change in arch/mips/net/ebpf_jit.c\n     - Drop change to bpf_jit_kallsyms\n     - Adjust filenames, context]\n    Signed-off-by: Ben Hutchings \u003cben.hutchings@codethink.co.uk\u003e\n    Signed-off-by: Greg Kroah-Hartman \u003cgregkh@linuxfoundation.org\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit ad818aeb27f38149ea00ff050b5aefb6c7dedb6c\nAuthor: Zi Shen Lim \u003czlim.lnx@gmail.com\u003e\nDate:   Wed Jun 8 21:18:48 2016 -0700\n\n    arm64: bpf: implement bpf_tail_call() helper\n\n    Add support for JMP_CALL_X (tail call) introduced by commit 04fd61ab36ec\n    (\"bpf: allow bpf programs to tail-call other bpf programs\").\n\n    bpf_tail_call() arguments:\n      ctx   - context pointer passed to next program\n      array - pointer to map which type is BPF_MAP_TYPE_PROG_ARRAY\n      index - index inside array that selects specific program to run\n\n    In this implementation arm64 JIT jumps into callee program after prologue,\n    so callee program reuses the same stack. For tail_call_cnt, we use the\n    callee-saved R26 (which was already saved/restored but previously unused\n    by JIT).\n\n    With this patch a tail call generates the following code on arm64:\n\n      if (index \u003e\u003d array-\u003emap.max_entries)\n          goto out;\n\n      34:   mov     x10, #0x10                      // #16\n      38:   ldr     w10, [x1,x10]\n      3c:   cmp     w2, w10\n      40:   b.ge    0x0000000000000074\n\n      if (tail_call_cnt \u003e MAX_TAIL_CALL_CNT)\n          goto out;\n      tail_call_cnt++;\n\n      44:   mov     x10, #0x20                      // #32\n      48:   cmp     x26, x10\n      4c:   b.gt    0x0000000000000074\n      50:   add     x26, x26, #0x1\n\n      prog \u003d array-\u003eptrs[index];\n      if (prog \u003d\u003d NULL)\n          goto out;\n\n      54:   mov     x10, #0x68                      // #104\n      58:   ldr     x10, [x1,x10]\n      5c:   ldr     x11, [x10,x2]\n      60:   cbz     x11, 0x0000000000000074\n\n      goto *(prog-\u003ebpf_func + prologue_size);\n\n      64:   mov     x10, #0x20                      // #32\n      68:   ldr     x10, [x11,x10]\n      6c:   add     x10, x10, #0x20\n      70:   br      x10\n      74:\n\n    Signed-off-by: Zi Shen Lim \u003czlim.lnx@gmail.com\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 14de6c12580533008343d6da357fc8285ce23a3c\nAuthor: Yang Shi \u003cyang.shi@linaro.org\u003e\nDate:   Mon May 16 16:36:26 2016 -0700\n\n    bpf: arm64: remove callee-save registers use for tmp registers\n\n    In the current implementation of ARM64 eBPF JIT, R23 and R24 are used for\n    tmp registers, which are callee-saved registers. This leads to variable size\n    of JIT prologue and epilogue. The latest blinding constant change prefers to\n    constant size of prologue and epilogue. AAPCS reserves R9 ~ R15 for temp\n    registers which not need to be saved/restored during function call. So, replace\n    R23 and R24 to R10 and R11, and remove tmp_used flag to save 2 instructions for\n    some jited BPF program.\n\n    CC: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Acked-by: Zi Shen Lim \u003czlim.lnx@gmail.com\u003e\n    Signed-off-by: Yang Shi \u003cyang.shi@linaro.org\u003e\n    Acked-by: Catalin Marinas \u003ccatalin.marinas@arm.com\u003e\n    Acked-by: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit cd6d8cc255e3396e4957cc452160fa4ac3b0188f\nAuthor: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\nDate:   Fri May 13 19:08:34 2016 +0200\n\n    bpf, arm64: add support for constant blinding\n\n    This patch adds recently added constant blinding helpers into the\n    arm64 eBPF JIT. In the bpf_int_jit_compile() path, requirements are\n    to utilize bpf_jit_blind_constants()/bpf_jit_prog_release_other()\n    pair for rewriting the program into a blinded one, and to map the\n    BPF_REG_AX register to a CPU register. The mapping is on x9.\n\n    Signed-off-by: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Acked-by: Zi Shen Lim \u003czlim.lnx@gmail.com\u003e\n    Acked-by: Yang Shi \u003cyang.shi@linaro.org\u003e\n    Tested-by: Yang Shi \u003cyang.shi@linaro.org\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit c8d6591e9094a942a2d6a1773fdd971343baf70c\nAuthor: Zi Shen Lim \u003czlim.lnx@gmail.com\u003e\nDate:   Wed Jan 13 23:33:22 2016 -0800\n\n    arm64: bpf: add extra pass to handle faulty codegen\n\n    Code generation functions in arch/arm64/kernel/insn.c previously\n    BUG_ON invalid parameters. Following change of that behavior, now we\n    need to handle the error case where AARCH64_BREAK_FAULT is returned.\n\n    Instead of error-handling on every emit() in JIT, we add a new\n    validation pass at the end of JIT compilation. There\u0027s no point in\n    running JITed code at run-time only to trap due to AARCH64_BREAK_FAULT.\n    Instead, we drop this failed JIT compilation and allow the system to\n    gracefully fallback on the BPF interpreter.\n\n    Signed-off-by: Zi Shen Lim \u003czlim.lnx@gmail.com\u003e\n    Suggested-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Acked-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 61a2502d89c0f7434196da6f413fb2ea2d0de5ba\nAuthor: Valdis Klētnieks \u003cvaldis.kletnieks@vt.edu\u003e\nDate:   Thu Jun 6 22:39:27 2019 -0400\n\n    bpf: silence warning messages in core\n\n    [ Upstream commit aee450cbe482a8c2f6fa5b05b178ef8b8ff107ca ]\n\n    Compiling kernel/bpf/core.c with W\u003d1 causes a flood of warnings:\n\n    kernel/bpf/core.c:1198:65: warning: initialized field overwritten [-Woverride-init]\n     1198 | #define BPF_INSN_3_TBL(x, y, z) [BPF_##x | BPF_##y | BPF_##z] \u003d true\n          |                                                                 ^~~~\n    kernel/bpf/core.c:1087:2: note: in expansion of macro \u0027BPF_INSN_3_TBL\u0027\n     1087 |  INSN_3(ALU, ADD,  X),   \\\n          |  ^~~~~~\n    kernel/bpf/core.c:1202:3: note: in expansion of macro \u0027BPF_INSN_MAP\u0027\n     1202 |   BPF_INSN_MAP(BPF_INSN_2_TBL, BPF_INSN_3_TBL),\n          |   ^~~~~~~~~~~~\n    kernel/bpf/core.c:1198:65: note: (near initialization for \u0027public_insntable[12]\u0027)\n     1198 | #define BPF_INSN_3_TBL(x, y, z) [BPF_##x | BPF_##y | BPF_##z] \u003d true\n          |                                                                 ^~~~\n    kernel/bpf/core.c:1087:2: note: in expansion of macro \u0027BPF_INSN_3_TBL\u0027\n     1087 |  INSN_3(ALU, ADD,  X),   \\\n          |  ^~~~~~\n    kernel/bpf/core.c:1202:3: note: in expansion of macro \u0027BPF_INSN_MAP\u0027\n     1202 |   BPF_INSN_MAP(BPF_INSN_2_TBL, BPF_INSN_3_TBL),\n          |   ^~~~~~~~~~~~\n\n    98 copies of the above.\n\n    The attached patch silences the warnings, because we *know* we\u0027re overwriting\n    the default initializer. That leaves bpf/core.c with only 6 other warnings,\n    which become more visible in comparison.\n\n    Signed-off-by: Valdis Kletnieks \u003cvaldis.kletnieks@vt.edu\u003e\n    Acked-by: Andrii Nakryiko \u003candriin@fb.com\u003e\n    Signed-off-by: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Signed-off-by: Sasha Levin \u003csashal@kernel.org\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 14ab33fe132b7e94832ded176597663e91b42c41\nAuthor: Chenbo Feng \u003cfengc@google.com\u003e\nDate:   Tue May 14 19:42:57 2019 -0700\n\n    UPSTREAM: bpf: relax inode permission check for retrieving bpf program\n\n    For iptable module to load a bpf program from a pinned location, it\n    only retrieve a loaded program and cannot change the program content so\n    requiring a write permission for it might not be necessary.\n    Also when adding or removing an unrelated iptable rule, it might need to\n    flush and reload the xt_bpf related rules as well and triggers the inode\n    permission check. It might be better to remove the write premission\n    check for the inode so we won\u0027t need to grant write access to all the\n    processes that flush and restore iptables rules.\n\n    Signed-off-by: Chenbo Feng \u003cfengc@google.com\u003e\n    Signed-off-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    (cherry picked from commit e547ff3f803e779a3898f1f48447b29f43c54085)\n\n    Bug: 129650054\n    Change-Id: I71487ad6f4d22e0a8be3757d9b72d1c04c92104d\n    (cherry picked from commit 9e74c1b9e8418aa0209b15db24f0b3d4876f52aa)\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit fa40b3355684c092ebdca700d04bd1de7d09e0c8\nAuthor: Alexei Starovoitov \u003cast@fb.com\u003e\nDate:   Thu May 9 19:33:54 2019 -0700\n\n    bpf: convert htab map to hlist_nulls\n\n    commit 4fe8435909fddc97b81472026aa954e06dd192a5 upstream.\n\n    when all map elements are pre-allocated one cpu can delete and reuse htab_elem\n    while another cpu is still walking the hlist. In such case the lookup may\n    miss the element. Convert hlist to hlist_nulls to avoid such scenario.\n    When bucket lock is taken there is no need to take such precautions,\n    so only convert map_lookup and map_get_next to nulls.\n    The race window is extremely small and only reproducible with explicit\n    udelay() inside lookup_nulls_elem_raw()\n\n    Similar to hlist add hlist_nulls_for_each_entry_safe() and\n    hlist_nulls_entry_safe() helpers.\n\n    Fixes: 6c9059817432 (\"bpf: pre-allocate hash map elements\")\n    Reported-by: Jonathan Perry \u003cjonperry@fb.com\u003e\n    Signed-off-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Acked-by: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Chenbo Feng \u003cfengc@google.com\u003e\n    Signed-off-by: Sasha Levin \u003csashal@kernel.org\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 0e39f5284b0a5c0424d1c446b5bd2e228cf69d86\nAuthor: Alexei Starovoitov \u003cast@fb.com\u003e\nDate:   Thu May 9 19:33:53 2019 -0700\n\n    bpf: fix struct htab_elem layout\n\n    commit 9f691549f76d488a0c74397b3e51e943865ea01f upstream.\n\n    when htab_elem is removed from the bucket list the htab_elem.hash_node.next\n    field should not be overridden too early otherwise we have a tiny race window\n    between lookup and delete.\n    The bug was discovered by manual code analysis and reproducible\n    only with explicit udelay() in lookup_elem_raw().\n\n    Fixes: 6c9059817432 (\"bpf: pre-allocate hash map elements\")\n    Reported-by: Jonathan Perry \u003cjonperry@fb.com\u003e\n    Signed-off-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Acked-by: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Chenbo Feng \u003cfengc@google.com\u003e\n    Signed-off-by: Sasha Levin \u003csashal@kernel.org\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit b2b555f2f28ff308ff373e78aeca0bb8c1a14ea9\nAuthor: Alexei Starovoitov \u003cast@kernel.org\u003e\nDate:   Mon Dec 3 22:46:04 2018 -0800\n\n    bpf: check pending signals while verifying programs\n\n    [ Upstream commit c3494801cd1785e2c25f1a5735fa19ddcf9665da ]\n\n    Malicious user space may try to force the verifier to use as much cpu\n    time and memory as possible. Hence check for pending signals\n    while verifying the program.\n    Note that suspend of sys_bpf(PROG_LOAD) syscall will lead to EAGAIN,\n    since the kernel has to release the resources used for program verification.\n\n    Reported-by: Anatoly Trosinenko \u003canatoly.trosinenko@gmail.com\u003e\n    Signed-off-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Acked-by: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Acked-by: Edward Cree \u003cecree@solarflare.com\u003e\n    Signed-off-by: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Signed-off-by: Sasha Levin \u003csashal@kernel.org\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 2e3c1d2b56458ea9d89ed6a4e2004b6771d9f779\nAuthor: Alexei Starovoitov \u003cast@kernel.org\u003e\nDate:   Tue May 15 09:27:05 2018 -0700\n\n    bpf: Prevent memory disambiguation attack\n\n    commit af86ca4e3088fe5eacf2f7e58c01fa68ca067672 upstream.\n\n    Detect code patterns where malicious \u0027speculative store bypass\u0027 can be used\n    and sanitize such patterns.\n\n     39: (bf) r3 \u003d r10\n     40: (07) r3 +\u003d -216\n     41: (79) r8 \u003d *(u64 *)(r7 +0)   // slow read\n     42: (7a) *(u64 *)(r10 -72) \u003d 0  // verifier inserts this instruction\n     43: (7b) *(u64 *)(r8 +0) \u003d r3   // this store becomes slow due to r8\n     44: (79) r1 \u003d *(u64 *)(r6 +0)   // cpu speculatively executes this load\n     45: (71) r2 \u003d *(u8 *)(r1 +0)    // speculatively arbitrary \u0027load byte\u0027\n                                     // is now sanitized\n\n    Above code after x86 JIT becomes:\n     e5: mov    %rbp,%rdx\n     e8: add    $0xffffffffffffff28,%rdx\n     ef: mov    0x0(%r13),%r14\n     f3: movq   $0x0,-0x48(%rbp)\n     fb: mov    %rdx,0x0(%r14)\n     ff: mov    0x0(%rbx),%rdi\n    103: movzbq 0x0(%rdi),%rsi\n\n    Signed-off-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Signed-off-by: Thomas Gleixner \u003ctglx@linutronix.de\u003e\n    [bwh: Backported to 4.9:\n     - Add bpf_verifier_env parameter to check_stack_write()\n     - Look up stack slot_types with state-\u003estack_slot_type[] rather than\n       state-\u003estack[].slot_type[]\n     - Drop bpf_verifier_env argument to verbose()\n     - Adjust context]\n    Signed-off-by: Ben Hutchings \u003cben.hutchings@codethink.co.uk\u003e\n    Signed-off-by: Greg Kroah-Hartman \u003cgregkh@linuxfoundation.org\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 30fa864349c6f77795e6ae6149752c0b3a724918\nAuthor: Ben Hutchings \u003cben.hutchings@codethink.co.uk\u003e\nDate:   Wed Dec 5 22:41:36 2018 +0000\n\n    bpf/verifier: Pass instruction index to check_mem_access() and check_xadd()\n\n    Extracted from commit 31fd85816dbe \"bpf: permits narrower load from\n    bpf program context fields\".\n\n    Cc: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Cc: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Signed-off-by: Ben Hutchings \u003cben.hutchings@codethink.co.uk\u003e\n    Signed-off-by: Greg Kroah-Hartman \u003cgregkh@linuxfoundation.org\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit c130ccf583d4f037e36666232aa804c4ac6fe41d\nAuthor: Ben Hutchings \u003cben.hutchings@codethink.co.uk\u003e\nDate:   Wed Dec 5 22:45:15 2018 +0000\n\n    bpf/verifier: Add spi variable to check_stack_write()\n\n    Extracted from commit dc503a8ad984 \"bpf/verifier: track liveness for\n    pruning\".\n\n    Cc: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Cc: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Signed-off-by: Ben Hutchings \u003cben.hutchings@codethink.co.uk\u003e\n    Signed-off-by: Greg Kroah-Hartman \u003cgregkh@linuxfoundation.org\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 2656b2b16ffb89c4d55e82e7de2f070555b9f836\nAuthor: Jakub Kicinski \u003cjakub.kicinski@netronome.com\u003e\nDate:   Thu May 3 18:37:17 2018 -0700\n\n    bpf: fix references to free_bpf_prog_info() in comments\n\n    [ Upstream commit ab7f5bf0928be2f148d000a6eaa6c0a36e74750e ]\n\n    Comments in the verifier refer to free_bpf_prog_info() which\n    seems to have never existed in tree.  Replace it with\n    free_used_maps().\n\n    Signed-off-by: Jakub Kicinski \u003cjakub.kicinski@netronome.com\u003e\n    Reviewed-by: Quentin Monnet \u003cquentin.monnet@netronome.com\u003e\n    Signed-off-by: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Signed-off-by: Sasha Levin \u003calexander.levin@microsoft.com\u003e\n    Signed-off-by: Greg Kroah-Hartman \u003cgregkh@linuxfoundation.org\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 425f0ce763baf8ca5a8120cd99348fa8c5633d83\nAuthor: Teng Qin \u003cqinteng@fb.com\u003e\nDate:   Mon Apr 24 19:00:37 2017 -0700\n\n    bpf: map_get_next_key to return first key on NULL\n\n    commit 8fe45924387be6b5c1be59a7eb330790c61d5d10 upstream.\n\n    When iterating through a map, we need to find a key that does not exist\n    in the map so map_get_next_key will give us the first key of the map.\n    This often requires a lot of guessing in production systems.\n\n    This patch makes map_get_next_key return the first key when the key\n    pointer in the parameter is NULL.\n\n    Signed-off-by: Teng Qin \u003cqinteng@fb.com\u003e\n    Signed-off-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Acked-by: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Chenbo Feng \u003cfengc@google.com\u003e\n    Cc: Lorenzo Colitti \u003clorenzo@google.com\u003e\n    Signed-off-by: Greg Kroah-Hartman \u003cgregkh@linuxfoundation.org\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit ebbedefd2db167856fea5315ecf0f0e4102631aa\nAuthor: Chenbo Feng \u003cfengc@google.com\u003e\nDate:   Mon Mar 19 17:57:27 2018 -0700\n\n    bpf: skip unnecessary capability check\n\n    commit 0fa4fe85f4724fff89b09741c437cbee9cf8b008 upstream.\n\n    The current check statement in BPF syscall will do a capability check\n    for CAP_SYS_ADMIN before checking sysctl_unprivileged_bpf_disabled. This\n    code path will trigger unnecessary security hooks on capability checking\n    and cause false alarms on unprivileged process trying to get CAP_SYS_ADMIN\n    access. This can be resolved by simply switch the order of the statement\n    and CAP_SYS_ADMIN is not required anyway if unprivileged bpf syscall is\n    allowed.\n\n    Signed-off-by: Chenbo Feng \u003cfengc@google.com\u003e\n    Acked-by: Lorenzo Colitti \u003clorenzo@google.com\u003e\n    Signed-off-by: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Signed-off-by: Greg Kroah-Hartman \u003cgregkh@linuxfoundation.org\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 27d0cafea1fd6570a6ca341d8dee178db6f985ad\nAuthor: Al Viro \u003cviro@zeniv.linux.org.uk\u003e\nDate:   Sat Dec 2 20:20:38 2017 -0500\n\n    BACKPORT: fix \"netfilter: xt_bpf: Fix XT_BPF_MODE_FD_PINNED mode of \u0027xt_bpf_info_v1\u0027\"\n\n    Descriptor table is a shared object; it\u0027s not a place where you can\n    stick temporary references to files, especially when we don\u0027t need\n    an opened file at all.\n\n    Cc: stable@vger.kernel.org # v4.14\n    Fixes: 98589a0998b8 (\"netfilter: xt_bpf: Fix XT_BPF_MODE_FD_PINNED mode of \u0027xt_bpf_info_v1\u0027\")\n    Signed-off-by: Al Viro \u003cviro@zeniv.linux.org.uk\u003e\n    Signed-off-by: Chenbo Feng \u003cfengc@google.com\u003e\n\n    Removed the code related to function bpf_prog_get_ok() since it is not\n    exsit in current android tree.\n    (cherry picked from commit 040ee69226f8a96b7943645d68f41d5d44b5ff7d)\n\n    Change-Id: If7a602128cdea4b4b50c8effb215c9bca7449515\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit eb13760b2d095d7c6cdc8db5c9ad71c110d90177\nAuthor: Jann Horn \u003cjannh@google.com\u003e\nDate:   Fri Dec 1 01:46:07 2017 +0100\n\n    UPSTREAM: netfilter: xt_bpf: add overflow checks\n\n    Check whether inputs from userspace are too long (explicit length field too\n    big or string not null-terminated) to avoid out-of-bounds reads.\n\n    As far as I can tell, this can at worst lead to very limited kernel heap\n    memory disclosure or oopses.\n\n    This bug can be triggered by an unprivileged user even if the xt_bpf module\n    is not loaded: iptables is available in network namespaces, and the xt_bpf\n    module can be autoloaded.\n\n    Triggering the bug with a classic BPF filter with fake length 0x1000 causes\n    the following KASAN report:\n\n    \u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\n    BUG: KASAN: slab-out-of-bounds in bpf_prog_create+0x84/0xf0\n    Read of size 32768 at addr ffff8801eff2c494 by task test/4627\n\n    CPU: 0 PID: 4627 Comm: test Not tainted 4.15.0-rc1+ #1\n    [...]\n    Call Trace:\n     dump_stack+0x5c/0x85\n     print_address_description+0x6a/0x260\n     kasan_report+0x254/0x370\n     ? bpf_prog_create+0x84/0xf0\n     memcpy+0x1f/0x50\n     bpf_prog_create+0x84/0xf0\n     bpf_mt_check+0x90/0xd6 [xt_bpf]\n    [...]\n    Allocated by task 4627:\n     kasan_kmalloc+0xa0/0xd0\n     __kmalloc_node+0x47/0x60\n     xt_alloc_table_info+0x41/0x70 [x_tables]\n    [...]\n    The buggy address belongs to the object at ffff8801eff2c3c0\n                    which belongs to the cache kmalloc-2048 of size 2048\n    The buggy address is located 212 bytes inside of\n                    2048-byte region [ffff8801eff2c3c0, ffff8801eff2cbc0)\n    [...]\n    \u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\n\n    Fixes: e6f30c731718 (\"netfilter: x_tables: add xt_bpf match\")\n    Signed-off-by: Jann Horn \u003cjannh@google.com\u003e\n    Signed-off-by: Pablo Neira Ayuso \u003cpablo@netfilter.org\u003e\n    Signed-off-by: Chenbo Feng \u003cfengc@google.com\u003e\n\n    (cherry picked from commit 6ab405114b0b229151ef06f4e31c7834dd09d0c0)\n\n    Change-Id: Ie066a9df84812853a9c9d2e51aa53646f4001542\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 195a85f0ceb365221f5d7845cfdda9591aaf1029\nAuthor: Shmulik Ladkani \u003cshmulik.ladkani@gmail.com\u003e\nDate:   Mon Oct 9 15:27:15 2017 +0300\n\n    UPSTREAM: netfilter: xt_bpf: Fix XT_BPF_MODE_FD_PINNED mode of \u0027xt_bpf_info_v1\u0027\n\n    Commit 2c16d6033264 (\"netfilter: xt_bpf: support ebpf\") introduced\n    support for attaching an eBPF object by an fd, with the\n    \u0027bpf_mt_check_v1\u0027 ABI expecting the \u0027.fd\u0027 to be specified upon each\n    IPT_SO_SET_REPLACE call.\n\n    However this breaks subsequent iptables calls:\n\n     # iptables -A INPUT -m bpf --object-pinned /sys/fs/bpf/xxx -j ACCEPT\n     # iptables -A INPUT -s 5.6.7.8 -j ACCEPT\n     iptables: Invalid argument. Run `dmesg\u0027 for more information.\n\n    That\u0027s because iptables works by loading existing rules using\n    IPT_SO_GET_ENTRIES to userspace, then issuing IPT_SO_SET_REPLACE with\n    the replacement set.\n\n    However, the loaded \u0027xt_bpf_info_v1\u0027 has an arbitrary \u0027.fd\u0027 number\n    (from the initial \"iptables -m bpf\" invocation) - so when 2nd invocation\n    occurs, userspace passes a bogus fd number, which leads to\n    \u0027bpf_mt_check_v1\u0027 to fail.\n\n    One suggested solution [1] was to hack iptables userspace, to perform a\n    \"entries fixup\" immediatley after IPT_SO_GET_ENTRIES, by opening a new,\n    process-local fd per every \u0027xt_bpf_info_v1\u0027 entry seen.\n\n    However, in [2] both Pablo Neira Ayuso and Willem de Bruijn suggested to\n    depricate the xt_bpf_info_v1 ABI dealing with pinned ebpf objects.\n\n    This fix changes the XT_BPF_MODE_FD_PINNED behavior to ignore the given\n    \u0027.fd\u0027 and instead perform an in-kernel lookup for the bpf object given\n    the provided \u0027.path\u0027.\n\n    It also defines an alias for the XT_BPF_MODE_FD_PINNED mode, named\n    XT_BPF_MODE_PATH_PINNED, to better reflect the fact that the user is\n    expected to provide the path of the pinned object.\n\n    Existing XT_BPF_MODE_FD_ELF behavior (non-pinned fd mode) is preserved.\n\n    References: [1] https://marc.info/?l\u003dnetfilter-devel\u0026m\u003d150564724607440\u0026w\u003d2\n                [2] https://marc.info/?l\u003dnetfilter-devel\u0026m\u003d150575727129880\u0026w\u003d2\n\n    Reported-by: Rafael Buchbinder \u003crafi@rbk.ms\u003e\n    Signed-off-by: Shmulik Ladkani \u003cshmulik.ladkani@gmail.com\u003e\n    Acked-by: Willem de Bruijn \u003cwillemb@google.com\u003e\n    Acked-by: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Signed-off-by: Pablo Neira Ayuso \u003cpablo@netfilter.org\u003e\n    Signed-off-by: Chenbo Feng \u003cfengc@google.com\u003e\n    (cherry picked from commit 98589a0998b8b13c4a8fa1ccb0e62751a019faa5)\n\n    Change-Id: Ia0d15a76823cca3afb38786a3d2c25c13ccf941d\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 4ba7b554f50c90a09d5ddc8bbdea6554906a5b1a\nAuthor: Willem de Bruijn \u003cwillemb@google.com\u003e\nDate:   Tue Dec 6 16:25:02 2016 -0500\n\n    UPSTREAM: netfilter: xt_bpf: support ebpf\n\n    Add support for attaching an eBPF object by file descriptor.\n\n    The iptables binary can be called with a path to an elf object or a\n    pinned bpf object. Also pass the mode and path to the kernel to be\n    able to return it later for iptables dump and save.\n\n    Signed-off-by: Willem de Bruijn \u003cwillemb@google.com\u003e\n    Signed-off-by: Pablo Neira Ayuso \u003cpablo@netfilter.org\u003e\n    Signed-off-by: Chenbo Feng \u003cfengc@google.com\u003e\n    (cherry picked from commit 2c16d60332643e90d4fa244f4a706c454b8c7569)\n\n    Change-Id: I31b8831a7ffd7c44985ee906ff194c1d934dafbe\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit ec42eb1f5122368db93dfcc53d6e6f56e26835cc\nAuthor: Alexei Starovoitov \u003cast@fb.com\u003e\nDate:   Thu Sep 1 18:37:24 2016 -0700\n\n    perf, bpf: add perf events core support for BPF_PROG_TYPE_PERF_EVENT programs\n\n    Allow attaching BPF_PROG_TYPE_PERF_EVENT programs to sw and hw perf events\n    via overflow_handler mechanism.\n    When program is attached the overflow_handlers become stacked.\n    The program acts as a filter.\n    Returning zero from the program means that the normal perf_event_output handler\n    will not be called and sampling event won\u0027t be stored in the ring buffer.\n\n    The overflow_handler_context\u003d\u003dNULL is an additional safety check\n    to make sure programs are not attached to hw breakpoints and watchdog\n    in case other checks (that prevent that now anyway) get accidentally\n    relaxed in the future.\n\n    The program refcnt is incremented in case perf_events are inhereted\n    when target task is forked.\n    Similar to kprobe and tracepoint programs there is no ioctl to\n    detach the program or swap already attached program. The user space\n    expected to close(perf_event_fd) like it does right now for kprobe+bpf.\n    That restriction simplifies the code quite a bit.\n\n    The invocation of overflow_handler in __perf_event_overflow() is now\n    done via READ_ONCE, since that pointer can be replaced when the program\n    is attached while perf_event itself could have been active already.\n    There is no need to do similar treatment for event-\u003eprog, since it\u0027s\n    assigned only once before it\u0027s accessed.\n\n    Signed-off-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit f3a8a6f4b22c6887e9fdf4522f86be6ef1e53f85\nAuthor: Wang Nan \u003cwangnan0@huawei.com\u003e\nDate:   Mon Mar 28 06:41:30 2016 +0000\n\n    perf/core: Set event\u0027s default ::overflow_handler()\n\n    Set a default event-\u003eoverflow_handler in perf_event_alloc() so don\u0027t\n    need to check event-\u003eoverflow_handler in __perf_event_overflow().\n    Following commits can give a different default overflow_handler.\n\n    Initial idea comes from Peter:\n\n      http://lkml.kernel.org/r/20130708121557.GA17211@twins.programming.kicks-ass.net\n\n    Since the default value of event-\u003eoverflow_handler is not NULL, existing\n    \u0027if (!overflow_handler)\u0027 checks need to be changed.\n\n    is_default_overflow_handler() is introduced for this.\n\n    No extra performance overhead is introduced into the hot path because in the\n    original code we still need to read this handler from memory. A conditional\n    branch is avoided so actually we remove some instructions.\n\n    Signed-off-by: Wang Nan \u003cwangnan0@huawei.com\u003e\n    Signed-off-by: Peter Zijlstra (Intel) \u003cpeterz@infradead.org\u003e\n    Cc: \u003cpi3orama@163.com\u003e\n    Cc: Alexander Shishkin \u003calexander.shishkin@linux.intel.com\u003e\n    Cc: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Cc: Arnaldo Carvalho de Melo \u003cacme@redhat.com\u003e\n    Cc: Brendan Gregg \u003cbrendan.d.gregg@gmail.com\u003e\n    Cc: He Kuang \u003chekuang@huawei.com\u003e\n    Cc: Jiri Olsa \u003cjolsa@kernel.org\u003e\n    Cc: Jiri Olsa \u003cjolsa@redhat.com\u003e\n    Cc: Linus Torvalds \u003ctorvalds@linux-foundation.org\u003e\n    Cc: Masami Hiramatsu \u003cmasami.hiramatsu.pt@hitachi.com\u003e\n    Cc: Namhyung Kim \u003cnamhyung@kernel.org\u003e\n    Cc: Peter Zijlstra \u003cpeterz@infradead.org\u003e\n    Cc: Stephane Eranian \u003ceranian@google.com\u003e\n    Cc: Thomas Gleixner \u003ctglx@linutronix.de\u003e\n    Cc: Vince Weaver \u003cvincent.weaver@maine.edu\u003e\n    Cc: Zefan Li \u003clizefan@huawei.com\u003e\n    Link: http://lkml.kernel.org/r/1459147292-239310-3-git-send-email-wangnan0@huawei.com\n    Signed-off-by: Ingo Molnar \u003cmingo@kernel.org\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 55be5db96c1bb48c44b5a2018e4c409b7d844b8b\nAuthor: Eric Dumazet \u003cedumazet@google.com\u003e\nDate:   Thu Mar 8 16:17:36 2018 +0100\n\n    bpf: add schedule points in percpu arrays management\n\n    [ upstream commit 32fff239de37ef226d5b66329dd133f64d63b22d ]\n\n    syszbot managed to trigger RCU detected stalls in\n    bpf_array_free_percpu()\n\n    It takes time to allocate a huge percpu map, but even more time to free\n    it.\n\n    Since we run in process context, use cond_resched() to yield cpu if\n    needed.\n\n    Fixes: a10423b87a7e (\"bpf: introduce BPF_MAP_TYPE_PERCPU_ARRAY map\")\n    Signed-off-by: Eric Dumazet \u003cedumazet@google.com\u003e\n    Reported-by: syzbot \u003csyzkaller@googlegroups.com\u003e\n    Signed-off-by: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Signed-off-by: Greg Kroah-Hartman \u003cgregkh@linuxfoundation.org\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit baf0ea99179eb7891d735a333370fbbc07f34db7\nAuthor: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\nDate:   Thu Mar 8 16:17:33 2018 +0100\n\n    bpf: fix mlock precharge on arraymaps\n\n    [ upstream commit 9c2d63b843a5c8a8d0559cc067b5398aa5ec3ffc ]\n\n    syzkaller recently triggered OOM during percpu map allocation;\n    while there is work in progress by Dennis Zhou to add __GFP_NORETRY\n    semantics for percpu allocator under pressure, there seems also a\n    missing bpf_map_precharge_memlock() check in array map allocation.\n\n    Given today the actual bpf_map_charge_memlock() happens after the\n    find_and_alloc_map() in syscall path, the bpf_map_precharge_memlock()\n    is there to bail out early before we go and do the map setup work\n    when we find that we hit the limits anyway. Therefore add this for\n    array map as well.\n\n    Fixes: 6c9059817432 (\"bpf: pre-allocate hash map elements\")\n    Fixes: a10423b87a7e (\"bpf: introduce BPF_MAP_TYPE_PERCPU_ARRAY map\")\n    Reported-by: syzbot+adb03f3f0bb57ce3acda@syzkaller.appspotmail.com\n    Signed-off-by: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Cc: Dennis Zhou \u003cdennisszhou@gmail.com\u003e\n    Signed-off-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Signed-off-by: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Signed-off-by: Greg Kroah-Hartman \u003cgregkh@linuxfoundation.org\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 2830ede9ba5ac6ebdca7a72e11fceb5711bd9d13\nAuthor: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\nDate:   Thu Mar 8 16:17:32 2018 +0100\n\n    bpf: fix wrong exposure of map_flags into fdinfo for lpm\n\n    [ upstream commit a316338cb71a3260201490e615f2f6d5c0d8fb2c ]\n\n    trie_alloc() always needs to have BPF_F_NO_PREALLOC passed in via\n    attr-\u003emap_flags, since it does not support preallocation yet. We\n    check the flag, but we never copy the flag into trie-\u003emap.map_flags,\n    which is later on exposed into fdinfo and used by loaders such as\n    iproute2. Latter uses this in bpf_map_selfcheck_pinned() to test\n    whether a pinned map has the same spec as the one from the BPF obj\n    file and if not, bails out, which is currently the case for lpm\n    since it exposes always 0 as flags.\n\n    Also copy over flags in array_map_alloc() and stack_map_alloc().\n    They always have to be 0 right now, but we should make sure to not\n    miss to copy them over at a later point in time when we add actual\n    flags for them to use.\n\n    Fixes: b95a5c4db09b (\"bpf: add a longest prefix match trie map implementation\")\n    Reported-by: Jarno Rajahalme \u003cjarno@covalent.io\u003e\n    Signed-off-by: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Acked-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Signed-off-by: Greg Kroah-Hartman \u003cgregkh@linuxfoundation.org\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 76afe46462fd92157dc93e4b4b896f673a81e6e0\nAuthor: Sami Tolvanen \u003csamitolvanen@google.com\u003e\nDate:   Thu Aug 24 08:59:31 2017 -0700\n\n    bpf: fix function type for __bpf_prog_run\n\n    Bug: 67506682\n    Change-Id: I096a470c65a2a1867c51da9a33843ae23bf5e547\n    Signed-off-by: Sami Tolvanen \u003csamitolvanen@google.com\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 31e0ee4629891c304a7367ed0eccadab700a0945\nAuthor: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\nDate:   Mon Jan 29 02:49:01 2018 +0100\n\n    bpf: reject stores into ctx via st and xadd\n\n    [ upstream commit f37a8cb84cce18762e8f86a70bd6a49a66ab964c ]\n\n    Alexei found that verifier does not reject stores into context\n    via BPF_ST instead of BPF_STX. And while looking at it, we\n    also should not allow XADD variant of BPF_STX.\n\n    The context rewriter is only assuming either BPF_LDX_MEM- or\n    BPF_STX_MEM-type operations, thus reject anything other than\n    that so that assumptions in the rewriter properly hold. Add\n    test cases as well for BPF selftests.\n\n    Fixes: d691f9e8d440 (\"bpf: allow programs to write to certain skb fields\")\n    Reported-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Signed-off-by: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Signed-off-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Signed-off-by: Greg Kroah-Hartman \u003cgregkh@linuxfoundation.org\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit fc8f593acb5fdfa686a5e4e3cd506ea23e81d1d2\nAuthor: Alexei Starovoitov \u003cast@kernel.org\u003e\nDate:   Mon Jan 29 02:49:00 2018 +0100\n\n    bpf: fix 32-bit divide by zero\n\n    [ upstream commit 68fda450a7df51cff9e5a4d4a4d9d0d5f2589153 ]\n\n    due to some JITs doing if (src_reg \u003d\u003d 0) check in 64-bit mode\n    for div/mod operations mask upper 32-bits of src register\n    before doing the check\n\n    Fixes: 622582786c9e (\"net: filter: x86: internal BPF JIT\")\n    Fixes: 7a12b5031c6b (\"sparc64: Add eBPF JIT.\")\n    Reported-by: syzbot+48340bb518e88849e2e3@syzkaller.appspotmail.com\n    Signed-off-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Signed-off-by: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Signed-off-by: Greg Kroah-Hartman \u003cgregkh@linuxfoundation.org\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 8379bad99cf4e0d63447c6c6b5cdc998328a7c15\nAuthor: Eric Dumazet \u003cedumazet@google.com\u003e\nDate:   Mon Jan 29 02:48:59 2018 +0100\n\n    bpf: fix divides by zero\n\n    [ upstream commit c366287ebd698ef5e3de300d90cd62ee9ee7373e ]\n\n    Divides by zero are not nice, lets avoid them if possible.\n\n    Also do_div() seems not needed when dealing with 32bit operands,\n    but this seems a minor detail.\n\n    Fixes: bd4cf0ed331a (\"net: filter: rework/optimize internal BPF interpreter\u0027s instruction set\")\n    Signed-off-by: Eric Dumazet \u003cedumazet@google.com\u003e\n    Reported-by: syzbot \u003csyzkaller@googlegroups.com\u003e\n    Signed-off-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Signed-off-by: Greg Kroah-Hartman \u003cgregkh@linuxfoundation.org\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 5469e46e82d4bec32a8752b8664569b571605ca3\nAuthor: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\nDate:   Mon Jan 29 02:48:57 2018 +0100\n\n    bpf: arsh is not supported in 32 bit alu thus reject it\n\n    [ upstream commit 7891a87efc7116590eaba57acc3c422487802c6f ]\n\n    The following snippet was throwing an \u0027unknown opcode cc\u0027 warning\n    in BPF interpreter:\n\n      0: (18) r0 \u003d 0x0\n      2: (7b) *(u64 *)(r10 -16) \u003d r0\n      3: (cc) (u32) r0 s\u003e\u003e\u003d (u32) r0\n      4: (95) exit\n\n    Although a number of JITs do support BPF_ALU | BPF_ARSH | BPF_{K,X}\n    generation, not all of them do and interpreter does neither. We can\n    leave existing ones and implement it later in bpf-next for the\n    remaining ones, but reject this properly in verifier for the time\n    being.\n\n    Fixes: 17a5267067f3 (\"bpf: verifier (add verifier core)\")\n    Reported-by: syzbot+93c4904c5c70348a6890@syzkaller.appspotmail.com\n    Signed-off-by: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Signed-off-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Signed-off-by: Greg Kroah-Hartman \u003cgregkh@linuxfoundation.org\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 6bfee5868d4e03e73d05d4024de45fc448d38411\nAuthor: Alexei Starovoitov \u003cast@kernel.org\u003e\nDate:   Mon Jan 29 02:48:56 2018 +0100\n\n    bpf: introduce BPF_JIT_ALWAYS_ON config\n\n    [ upstream commit 290af86629b25ffd1ed6232c4e9107da031705cb ]\n\n    The BPF interpreter has been used as part of the spectre 2 attack CVE-2017-5715.\n\n    A quote from goolge project zero blog:\n    \"At this point, it would normally be necessary to locate gadgets in\n    the host kernel code that can be used to actually leak data by reading\n    from an attacker-controlled location, shifting and masking the result\n    appropriately and then using the result of that as offset to an\n    attacker-controlled address for a load. But piecing gadgets together\n    and figuring out which ones work in a speculation context seems annoying.\n    So instead, we decided to use the eBPF interpreter, which is built into\n    the host kernel - while there is no legitimate way to invoke it from inside\n    a VM, the presence of the code in the host kernel\u0027s text section is sufficient\n    to make it usable for the attack, just like with ordinary ROP gadgets.\"\n\n    To make attacker job harder introduce BPF_JIT_ALWAYS_ON config\n    option that removes interpreter from the kernel in favor of JIT-only mode.\n    So far eBPF JIT is supported by:\n    x64, arm64, arm32, sparc64, s390, powerpc64, mips64\n\n    The start of JITed program is randomized and code page is marked as read-only.\n    In addition \"constant blinding\" can be turned on with net.core.bpf_jit_harden\n\n    v2-\u003ev3:\n    - move __bpf_prog_ret0 under ifdef (Daniel)\n\n    v1-\u003ev2:\n    - fix init order, test_bpf and cBPF (Daniel\u0027s feedback)\n    - fix offloaded bpf (Jakub\u0027s feedback)\n    - add \u0027return 0\u0027 dummy in case something can invoke prog-\u003ebpf_func\n    - retarget bpf tree. For bpf-next the patch would need one extra hunk.\n      It will be sent when the trees are merged back to net-next\n\n    Considered doing:\n      int bpf_jit_enable __read_mostly \u003d BPF_EBPF_JIT_DEFAULT;\n    but it seems better to land the patch as-is and in bpf-next remove\n    bpf_jit_enable global variable from all JITs, consolidate in one place\n    and remove this jit_init() function.\n\n    Signed-off-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Signed-off-by: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Signed-off-by: Greg Kroah-Hartman \u003cgregkh@linuxfoundation.org\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 4fcbe880a7f70d56508dc6d2c5748dcb6b6417ba\nAuthor: Alexei Starovoitov \u003cast@fb.com\u003e\nDate:   Mon Jan 29 02:48:55 2018 +0100\n\n    bpf: fix bpf_tail_call() x64 JIT\n\n    [ upstream commit 90caccdd8cc0215705f18b92771b449b01e2474a ]\n\n    - bpf prog_array just like all other types of bpf array accepts 32-bit index.\n      Clarify that in the comment.\n    - fix x64 JIT of bpf_tail_call which was incorrectly loading 8 instead of 4 bytes\n    - tighten corresponding check in the interpreter to stay consistent\n\n    The JIT bug can be triggered after introduction of BPF_F_NUMA_NODE flag\n    in commit 96eabe7a40aa in 4.14. Before that the map_flags would stay zero and\n    though JIT code is wrong it will check bounds correctly.\n    Hence two fixes tags. All other JITs don\u0027t have this problem.\n\n    Signed-off-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Fixes: 96eabe7a40aa (\"bpf: Allow selecting numa node during map creation\")\n    Fixes: b52f00e6a715 (\"x86: bpf_jit: implement bpf_tail_call() helper\")\n    Acked-by: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Acked-by: Martin KaFai Lau \u003ckafai@fb.com\u003e\n    Reviewed-by: Eric Dumazet \u003cedumazet@google.com\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Greg Kroah-Hartman \u003cgregkh@linuxfoundation.org\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit a3fd3f0275a74f5ff9bfc37b767fe5abc50b647f\nAuthor: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\nDate:   Wed Jan 10 23:25:05 2018 +0100\n\n    bpf, array: fix overflow in max_entries and undefined behavior in index_mask\n\n    commit bbeb6e4323dad9b5e0ee9f60c223dd532e2403b1 upstream.\n\n    syzkaller tried to alloc a map with 0xfffffffd entries out of a userns,\n    and thus unprivileged. With the recently added logic in b2157399cc98\n    (\"bpf: prevent out-of-bounds speculation\") we round this up to the next\n    power of two value for max_entries for unprivileged such that we can\n    apply proper masking into potentially zeroed out map slots.\n\n    However, this will generate an index_mask of 0xffffffff, and therefore\n    a + 1 will let this overflow into new max_entries of 0. This will pass\n    allocation, etc, and later on map access we still enforce on the original\n    attr-\u003emax_entries value which was 0xfffffffd, therefore triggering GPF\n    all over the place. Thus bail out on overflow in such case.\n\n    Moreover, on 32 bit archs roundup_pow_of_two() can also not be used,\n    since fls_long(max_entries - 1) can result in 32 and 1UL \u003c\u003c 32 in 32 bit\n    space is undefined. Therefore, do this by hand in a 64 bit variable.\n\n    This fixes all the issues triggered by syzkaller\u0027s reproducers.\n\n    Fixes: b2157399cc98 (\"bpf: prevent out-of-bounds speculation\")\n    Reported-by: syzbot+b0efb8e572d01bce1ae0@syzkaller.appspotmail.com\n    Reported-by: syzbot+6c15e9744f75f2364773@syzkaller.appspotmail.com\n    Reported-by: syzbot+d2f5524fb46fd3b312ee@syzkaller.appspotmail.com\n    Reported-by: syzbot+61d23c95395cc90dbc2b@syzkaller.appspotmail.com\n    Reported-by: syzbot+0d363c942452cca68c01@syzkaller.appspotmail.com\n    Signed-off-by: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Signed-off-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Signed-off-by: Greg Kroah-Hartman \u003cgregkh@linuxfoundation.org\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 46583a2655df94b4cf55bddaed7e7e1e3a9c9022\nAuthor: Alexei Starovoitov \u003cast@kernel.org\u003e\nDate:   Sun Jan 7 17:33:02 2018 -0800\n\n    bpf: prevent out-of-bounds speculation\n\n    commit b2157399cc9898260d6031c5bfe45fe137c1fbe7 upstream.\n\n    Under speculation, CPUs may mis-predict branches in bounds checks. Thus,\n    memory accesses under a bounds check may be speculated even if the\n    bounds check fails, providing a primitive for building a side channel.\n\n    To avoid leaking kernel data round up array-based maps and mask the index\n    after bounds check, so speculated load with out of bounds index will load\n    either valid value from the array or zero from the padded area.\n\n    Unconditionally mask index for all array types even when max_entries\n    are not rounded to power of 2 for root user.\n    When map is created by unpriv user generate a sequence of bpf insns\n    that includes AND operation to make sure that JITed code includes\n    the same \u0027index \u0026 index_mask\u0027 operation.\n\n    If prog_array map is created by unpriv user replace\n      bpf_tail_call(ctx, map, index);\n    with\n      if (index \u003e\u003d max_entries) {\n        index \u0026\u003d map-\u003eindex_mask;\n        bpf_tail_call(ctx, map, index);\n      }\n    (along with roundup to power 2) to prevent out-of-bounds speculation.\n    There is secondary redundant \u0027if (index \u003e\u003d max_entries)\u0027 in the interpreter\n    and in all JITs, but they can be optimized later if necessary.\n\n    Other array-like maps (cpumap, devmap, sockmap, perf_event_array, cgroup_array)\n    cannot be used by unpriv, so no changes there.\n\n    That fixes bpf side of \"Variant 1: bounds check bypass (CVE-2017-5753)\" on\n    all architectures with and without JIT.\n\n    v2-\u003ev3:\n    Daniel noticed that attack potentially can be crafted via syscall commands\n    without loading the program, so add masking to those paths as well.\n\n    Signed-off-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Acked-by: John Fastabend \u003cjohn.fastabend@gmail.com\u003e\n    Signed-off-by: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Cc: Jiri Slaby \u003cjslaby@suse.cz\u003e\n    [ Backported to 4.9 - gregkh ]\n    Signed-off-by: Greg Kroah-Hartman \u003cgregkh@linuxfoundation.org\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 2a398d547a76faffcb7666c38341f0c166a4f919\nAuthor: Alexei Starovoitov \u003cast@fb.com\u003e\nDate:   Wed Mar 15 18:26:40 2017 -0700\n\n    bpf: refactor fixup_bpf_calls()\n\n    commit 79741b3bdec01a8628368fbcfccc7d189ed606cb upstream.\n\n    reduce indent and make it iterate over instructions similar to\n    convert_ctx_accesses(). Also convert hard BUG_ON into soft verifier error.\n\n    Signed-off-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Acked-by: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Cc: Jiri Slaby \u003cjslaby@suse.cz\u003e\n    [Backported to 4.9.y - gregkh]\n    Signed-off-by: Greg Kroah-Hartman \u003cgregkh@linuxfoundation.org\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit ca289fbdf612b7419fef48957bb6a5bc6c94e937\nAuthor: Alexei Starovoitov \u003cast@fb.com\u003e\nDate:   Wed Mar 15 18:26:39 2017 -0700\n\n    bpf: move fixup_bpf_calls() function\n\n    commit e245c5c6a5656e4d61aa7bb08e9694fd6e5b2b9d upstream.\n\n    no functional change.\n    move fixup_bpf_calls() to verifier.c\n    it\u0027s being refactored in the next patch\n\n    Signed-off-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Acked-by: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Cc: Jiri Slaby \u003cjslaby@suse.cz\u003e\n    [backported to 4.9 - gregkh]\n    Signed-off-by: Greg Kroah-Hartman \u003cgregkh@linuxfoundation.org\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit e06abe187875e684438cdd66167cfc1d3cfae4f7\nAuthor: Ben Hutchings \u003cben@decadent.org.uk\u003e\nDate:   Sat Dec 23 02:26:17 2017 +0000\n\n    bpf/verifier: Fix states_equal() comparison of pointer and UNKNOWN\n\n    An UNKNOWN_VALUE is not supposed to be derived from a pointer, unless\n    pointer leaks are allowed.  Therefore, states_equal() must not treat\n    a state with a pointer in a register as \"equal\" to a state with an\n    UNKNOWN_VALUE in that register.\n\n    This was fixed differently upstream, but the code around here was\n    largely rewritten in 4.14 by commit f1174f77b50c \"bpf/verifier: rework\n    value tracking\".  The bug can be detected by the bpf/verifier sub-test\n    \"pointer/scalar confusion in state equality check (way 1)\".\n\n    Signed-off-by: Ben Hutchings \u003cben@decadent.org.uk\u003e\n    Cc: Edward Cree \u003cecree@solarflare.com\u003e\n    Cc: Jann Horn \u003cjannh@google.com\u003e\n    Cc: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Cc: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit e7f00ca3c36a454254a676147a0469ca33a58dd9\nAuthor: Jann Horn \u003cjannh@google.com\u003e\nDate:   Fri Dec 22 16:29:05 2017 +0100\n\n    bpf: fix incorrect sign extension in check_alu_op()\n\n    [ Upstream commit 95a762e2c8c942780948091f8f2a4f32fce1ac6f ]\n\n    Distinguish between\n    BPF_ALU64|BPF_MOV|BPF_K (load 32-bit immediate, sign-extended to 64-bit)\n    and BPF_ALU|BPF_MOV|BPF_K (load 32-bit immediate, zero-padded to 64-bit);\n    only perform sign extension in the first case.\n\n    Starting with v4.14, this is exploitable by unprivileged users as long as\n    the unprivileged_bpf_disabled sysctl isn\u0027t set.\n\n    Debian assigned CVE-2017-16995 for this issue.\n\n    v3:\n     - add CVE number (Ben Hutchings)\n\n    Fixes: 484611357c19 (\"bpf: allow access into map value arrays\")\n    Signed-off-by: Jann Horn \u003cjannh@google.com\u003e\n    Acked-by: Edward Cree \u003cecree@solarflare.com\u003e\n    Signed-off-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Signed-off-by: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Signed-off-by: Greg Kroah-Hartman \u003cgregkh@linuxfoundation.org\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit ad060c2d83c42fab8c3e8615f1a178c09dd28cb5\nAuthor: Jann Horn \u003cjannh@google.com\u003e\nDate:   Fri Dec 22 16:29:04 2017 +0100\n\n    bpf: reject out-of-bounds stack pointer calculation\n\n    Reject programs that compute wildly out-of-bounds stack pointers.\n    Otherwise, pointers can be computed with an offset that doesn\u0027t fit into an\n    `int`, causing security issues in the stack memory access check (as well as\n    signed integer overflow during offset addition).\n\n    This is a fix specifically for the v4.9 stable tree because the mainline\n    code looks very different at this point.\n\n    Fixes: 7bca0a9702edf (\"bpf: enhance verifier to understand stack pointer arithmetic\")\n    Signed-off-by: Jann Horn \u003cjannh@google.com\u003e\n    Acked-by: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Signed-off-by: Greg Kroah-Hartman \u003cgregkh@linuxfoundation.org\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 5fc649fb72af20b335ff0b51589edf45e25189da\nAuthor: Alexei Starovoitov \u003cast@fb.com\u003e\nDate:   Fri Dec 22 16:29:03 2017 +0100\n\n    bpf: fix branch pruning logic\n\n    [ Upstream commit c131187db2d3fa2f8bf32fdf4e9a4ef805168467 ]\n\n    when the verifier detects that register contains a runtime constant\n    and it\u0027s compared with another constant it will prune exploration\n    of the branch that is guaranteed not to be taken at runtime.\n    This is all correct, but malicious program may be constructed\n    in such a way that it always has a constant comparison and\n    the other branch is never taken under any conditions.\n    In this case such path through the program will not be explored\n    by the verifier. It won\u0027t be taken at run-time either, but since\n    all instructions are JITed the malicious program may cause JITs\n    to complain about using reserved fields, etc.\n    To fix the issue we have to track the instructions explored by\n    the verifier and sanitize instructions that are dead at run time\n    with NOPs. We cannot reject such dead code, since llvm generates\n    it for valid C code, since it doesn\u0027t do as much data flow\n    analysis as the verifier does.\n\n    Fixes: 17a5267067f3 (\"bpf: verifier (add verifier core)\")\n    Signed-off-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Acked-by: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Signed-off-by: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Signed-off-by: Greg Kroah-Hartman \u003cgregkh@linuxfoundation.org\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 1c4b04339ed65e9212af21d97ea8492959241bc6\nAuthor: Alexei Starovoitov \u003cast@fb.com\u003e\nDate:   Fri Dec 22 16:29:02 2017 +0100\n\n    bpf: adjust insn_aux_data when patching insns\n\n    [ Upstream commit 8041902dae5299c1f194ba42d14383f734631009 ]\n\n    convert_ctx_accesses() replaces single bpf instruction with a set of\n    instructions. Adjust corresponding insn_aux_data while patching.\n    It\u0027s needed to make sure subsequent \u0027for(all insn)\u0027 loops\n    have matching insn and insn_aux_data.\n\n    Signed-off-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Acked-by: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Greg Kroah-Hartman \u003cgregkh@linuxfoundation.org\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 4b581372fe405e6474a914502e9abf3a9f8295d4\nAuthor: Eric Dumazet \u003cedumazet@google.com\u003e\nDate:   Tue Nov 14 17:15:50 2017 -0800\n\n    bpf: fix lockdep splat\n\n    [ Upstream commit 89ad2fa3f043a1e8daae193bcb5fe34d5f8caf28 ]\n\n    pcpu_freelist_pop() needs the same lockdep awareness than\n    pcpu_freelist_populate() to avoid a false positive.\n\n     [ INFO: SOFTIRQ-safe -\u003e SOFTIRQ-unsafe lock order detected ]\n\n     switchto-defaul/12508 [HC0[0]:SC0[6]:HE0:SE0] is trying to acquire:\n      (\u0026htab-\u003ebuckets[i].lock){......}, at: [\u003cffffffff9dc099cb\u003e] __htab_percpu_map_update_elem+0x1cb/0x300\n\n     and this task is already holding:\n      (dev_queue-\u003edev-\u003eqdisc_class ?: \u0026qdisc_tx_lock#2){+.-...}, at: [\u003cffffffff9e135848\u003e] __dev_queue_xmit+0\n    x868/0x1240\n     which would create a new lock dependency:\n      (dev_queue-\u003edev-\u003eqdisc_class ?: \u0026qdisc_tx_lock#2){+.-...} -\u003e (\u0026htab-\u003ebuckets[i].lock){......}\n\n     but this new dependency connects a SOFTIRQ-irq-safe lock:\n      (dev_queue-\u003edev-\u003eqdisc_class ?: \u0026qdisc_tx_lock#2){+.-...}\n     ... which became SOFTIRQ-irq-safe at:\n       [\u003cffffffff9db5931b\u003e] __lock_acquire+0x42b/0x1f10\n       [\u003cffffffff9db5b32c\u003e] lock_acquire+0xbc/0x1b0\n       [\u003cffffffff9da05e38\u003e] _raw_spin_lock+0x38/0x50\n       [\u003cffffffff9e135848\u003e] __dev_queue_xmit+0x868/0x1240\n       [\u003cffffffff9e136240\u003e] dev_queue_xmit+0x10/0x20\n       [\u003cffffffff9e1965d9\u003e] ip_finish_output2+0x439/0x590\n       [\u003cffffffff9e197410\u003e] ip_finish_output+0x150/0x2f0\n       [\u003cffffffff9e19886d\u003e] ip_output+0x7d/0x260\n       [\u003cffffffff9e19789e\u003e] ip_local_out+0x5e/0xe0\n       [\u003cffffffff9e197b25\u003e] ip_queue_xmit+0x205/0x620\n       [\u003cffffffff9e1b8398\u003e] tcp_transmit_skb+0x5a8/0xcb0\n       [\u003cffffffff9e1ba152\u003e] tcp_write_xmit+0x242/0x1070\n       [\u003cffffffff9e1baffc\u003e] __tcp_push_pending_frames+0x3c/0xf0\n       [\u003cffffffff9e1b3472\u003e] tcp_rcv_established+0x312/0x700\n       [\u003cffffffff9e1c1acc\u003e] tcp_v4_do_rcv+0x11c/0x200\n       [\u003cffffffff9e1c3dc2\u003e] tcp_v4_rcv+0xaa2/0xc30\n       [\u003cffffffff9e191107\u003e] ip_local_deliver_finish+0xa7/0x240\n       [\u003cffffffff9e191a36\u003e] ip_local_deliver+0x66/0x200\n       [\u003cffffffff9e19137d\u003e] ip_rcv_finish+0xdd/0x560\n       [\u003cffffffff9e191e65\u003e] ip_rcv+0x295/0x510\n       [\u003cffffffff9e12ff88\u003e] __netif_receive_skb_core+0x988/0x1020\n       [\u003cffffffff9e130641\u003e] __netif_receive_skb+0x21/0x70\n       [\u003cffffffff9e1306ff\u003e] process_backlog+0x6f/0x230\n       [\u003cffffffff9e132129\u003e] net_rx_action+0x229/0x420\n       [\u003cffffffff9da07ee8\u003e] __do_softirq+0xd8/0x43d\n       [\u003cffffffff9e282bcc\u003e] do_softirq_own_stack+0x1c/0x30\n       [\u003cffffffff9dafc2f5\u003e] do_softirq+0x55/0x60\n       [\u003cffffffff9dafc3a8\u003e] __local_bh_enable_ip+0xa8/0xb0\n       [\u003cffffffff9db4c727\u003e] cpu_startup_entry+0x1c7/0x500\n       [\u003cffffffff9daab333\u003e] start_secondary+0x113/0x140\n\n     to a SOFTIRQ-irq-unsafe lock:\n      (\u0026head-\u003elock){+.+...}\n     ... which became SOFTIRQ-irq-unsafe at:\n     ...  [\u003cffffffff9db5971f\u003e] __lock_acquire+0x82f/0x1f10\n       [\u003cffffffff9db5b32c\u003e] lock_acquire+0xbc/0x1b0\n       [\u003cffffffff9da05e38\u003e] _raw_spin_lock+0x38/0x50\n       [\u003cffffffff9dc0b7fa\u003e] pcpu_freelist_pop+0x7a/0xb0\n       [\u003cffffffff9dc08b2c\u003e] htab_map_alloc+0x50c/0x5f0\n       [\u003cffffffff9dc00dc5\u003e] SyS_bpf+0x265/0x1200\n       [\u003cffffffff9e28195f\u003e] entry_SYSCALL_64_fastpath+0x12/0x17\n\n     other info that might help us debug this:\n\n     Chain exists of:\n       dev_queue-\u003edev-\u003eqdisc_class ?: \u0026qdisc_tx_lock#2 --\u003e \u0026htab-\u003ebuckets[i].lock --\u003e \u0026head-\u003elock\n\n      Possible interrupt unsafe locking scenario:\n\n            CPU0                    CPU1\n            ----                    ----\n       lock(\u0026head-\u003elock);\n                                    local_irq_disable();\n                                    lock(dev_queue-\u003edev-\u003eqdisc_class ?: \u0026qdisc_tx_lock#2);\n                                    lock(\u0026htab-\u003ebuckets[i].lock);\n       \u003cInterrupt\u003e\n         lock(dev_queue-\u003edev-\u003eqdisc_class ?: \u0026qdisc_tx_lock#2);\n\n      *** DEADLOCK ***\n\n    Fixes: e19494edab82 (\"bpf: introduce percpu_freelist\")\n    Signed-off-by: Eric Dumazet \u003cedumazet@google.com\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Sasha Levin \u003calexander.levin@verizon.com\u003e\n    Signed-off-by: Greg Kroah-Hartman \u003cgregkh@linuxfoundation.org\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 42cd8952e743e3b356722237b110ed39cd49acdf\nAuthor: Chenbo Feng \u003cfengc@google.com\u003e\nDate:   Wed Oct 18 13:00:26 2017 -0700\n\n    UPSTREAM: selinux: bpf: Add addtional check for bpf object file receive\n\n    Introduce a bpf object related check when sending and receiving files\n    through unix domain socket as well as binder. It checks if the receiving\n    process have privilege to read/write the bpf map or use the bpf program.\n    This check is necessary because the bpf maps and programs are using a\n    anonymous inode as their shared inode so the normal way of checking the\n    files and sockets when passing between processes cannot work properly on\n    eBPF object. This check only works when the BPF_SYSCALL is configured.\n\n    Signed-off-by: Chenbo Feng \u003cfengc@google.com\u003e\n    Acked-by: Stephen Smalley \u003csds@tycho.nsa.gov\u003e\n    Reviewed-by: James Morris \u003cjames.l.morris@oracle.com\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n\n    (cherry-pick from net-next: f66e448cfda021b0bcd884f26709796fe19c7cc1)\n    Bug: 30950746\n\n    Change-Id: I5b2cf4ccb4eab7eda91ddd7091d6aa3e7ed9f2cd\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit b760311edd0c8918efddc8161b4ba1127cc00cd8\nAuthor: Chenbo Feng \u003cfengc@google.com\u003e\nDate:   Wed Oct 18 13:00:25 2017 -0700\n\n    UPSTREAM: selinux: bpf: Add selinux check for eBPF syscall operations\n\n    Implement the actual checks introduced to eBPF related syscalls. This\n    implementation use the security field inside bpf object to store a sid that\n    identify the bpf object. And when processes try to access the object,\n    selinux will check if processes have the right privileges. The creation\n    of eBPF object are also checked at the general bpf check hook and new\n    cmd introduced to eBPF domain can also be checked there.\n\n    Signed-off-by: Chenbo Feng \u003cfengc@google.com\u003e\n    Acked-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Reviewed-by: James Morris \u003cjames.l.morris@oracle.com\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n\n    (cherry-pick from net-next: ec27c3568a34c7fe5fcf4ac0a354eda77687f7eb)\n    Bug: 30950746\n    Change-Id: Ifb0cdd4b7d470223b143646b339ba511ac77c156\n\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n    Change-Id: If49bc4c91c152efd36b372f8dfa15486e916f7df\n\ncommit 54b197e92fb0da064e7425a6646a52543e5071ee\nAuthor: Chenbo Feng \u003cfengc@google.com\u003e\nDate:   Wed Oct 18 13:00:24 2017 -0700\n\n    BACKPORT: security: bpf: Add LSM hooks for bpf object related syscall\n\n    Introduce several LSM hooks for the syscalls that will allow the\n    userspace to access to eBPF object such as eBPF programs and eBPF maps.\n    The security check is aimed to enforce a per object security protection\n    for eBPF object so only processes with the right priviliges can\n    read/write to a specific map or use a specific eBPF program. Besides\n    that, a general security hook is added before the multiplexer of bpf\n    syscall to check the cmd and the attribute used for the command. The\n    actual security module can decide which command need to be checked and\n    how the cmd should be checked.\n\n    Signed-off-by: Chenbo Feng \u003cfengc@google.com\u003e\n    Acked-by: James Morris \u003cjames.l.morris@oracle.com\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n\n    Added the LIST_HEAD_INIT call for security hooks, it nolonger exist in\n    uptream code.\n    (cherry-pick from net-next: afdb09c720b62b8090584c11151d856df330e57d)\n    Bug: 30950746\n\n    Change-Id: Ieb3ac74392f531735fc7c949b83346a5f587a77b\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 636d72b2d2ab8a78ada6f2ec48e9ecd83a3479ab\nAuthor: Chenbo Feng \u003cfengc@google.com\u003e\nDate:   Wed Oct 18 13:00:22 2017 -0700\n\n    BACKPORT: bpf: Add file mode configuration into bpf maps\n\n    Introduce the map read/write flags to the eBPF syscalls that returns the\n    map fd. The flags is used to set up the file mode when construct a new\n    file descriptor for bpf maps. To not break the backward capability, the\n    f_flags is set to O_RDWR if the flag passed by syscall is 0. Otherwise\n    it should be O_RDONLY or O_WRONLY. When the userspace want to modify or\n    read the map content, it will check the file mode to see if it is\n    allowed to make the change.\n\n    Signed-off-by: Chenbo Feng \u003cfengc@google.com\u003e\n    Acked-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Acked-by: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n\n    Deleted the file mode configuration code in unsupported map type and\n    removed the file mode check in non-existing helper functions.\n    (cherry-pick from net-next: 6e71b04a82248ccf13a94b85cbc674a9fefe53f5)\n    Bug: 30950746\n\n    Change-Id: Icfad20f1abb77f91068d244fb0d87fa40824dd1b\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit e9abb113a304794156cb9e7d96755578ef098f62\nAuthor: Anay Wadhera \u003cawadhera@berkeley.edu\u003e\nDate:   Fri May 21 13:29:12 2021 -0700\n\n    bpf: move bpf_map_show_fdinfo to match upstream location\n\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 68e67ead30cd7d656604eeea3d793c33d7f45f0f\nAuthor: Edward Cree \u003cecree@solarflare.com\u003e\nDate:   Fri Sep 15 14:37:38 2017 +0100\n\n    bpf/verifier: reject BPF_ALU64|BPF_END\n\n    [ Upstream commit e67b8a685c7c984e834e3181ef4619cd7025a136 ]\n\n    Neither ___bpf_prog_run nor the JITs accept it.\n    Also adds a new test case.\n\n    Fixes: 17a5267067f3 (\"bpf: verifier (add verifier core)\")\n    Signed-off-by: Edward Cree \u003cecree@solarflare.com\u003e\n    Acked-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Acked-by: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Greg Kroah-Hartman \u003cgregkh@linuxfoundation.org\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 10272c98adaafe9cfbac01e5e906e8b93eb41df2\nAuthor: Edward Cree \u003cecree@solarflare.com\u003e\nDate:   Fri Jul 21 14:37:34 2017 +0100\n\n    bpf/verifier: fix min/max handling in BPF_SUB\n\n    [ Upstream commit 9305706c2e808ae59f1eb201867f82f1ddf6d7a6 ]\n\n    We have to subtract the src max from the dst min, and vice-versa, since\n     (e.g.) the smallest result comes from the largest subtrahend.\n\n    Fixes: 484611357c19 (\"bpf: allow access into map value arrays\")\n    Signed-off-by: Edward Cree \u003cecree@solarflare.com\u003e\n    Acked-by: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Greg Kroah-Hartman \u003cgregkh@linuxfoundation.org\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit b56743acc363807bc6f9f4e54409debe32ff0d74\nAuthor: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\nDate:   Fri Jul 21 00:00:21 2017 +0200\n\n    bpf: fix mixed signed/unsigned derived min/max value bounds\n\n    [ Upstream commit 4cabc5b186b5427b9ee5a7495172542af105f02b ]\n\n    Edward reported that there\u0027s an issue in min/max value bounds\n    tracking when signed and unsigned compares both provide hints\n    on limits when having unknown variables. E.g. a program such\n    as the following should have been rejected:\n\n       0: (7a) *(u64 *)(r10 -8) \u003d 0\n       1: (bf) r2 \u003d r10\n       2: (07) r2 +\u003d -8\n       3: (18) r1 \u003d 0xffff8a94cda93400\n       5: (85) call bpf_map_lookup_elem#1\n       6: (15) if r0 \u003d\u003d 0x0 goto pc+7\n      R0\u003dmap_value(ks\u003d8,vs\u003d8,id\u003d0),min_value\u003d0,max_value\u003d0 R10\u003dfp\n       7: (7a) *(u64 *)(r10 -16) \u003d -8\n       8: (79) r1 \u003d *(u64 *)(r10 -16)\n       9: (b7) r2 \u003d -1\n      10: (2d) if r1 \u003e r2 goto pc+3\n      R0\u003dmap_value(ks\u003d8,vs\u003d8,id\u003d0),min_value\u003d0,max_value\u003d0 R1\u003dinv,min_value\u003d0\n      R2\u003dimm-1,max_value\u003d18446744073709551615,min_align\u003d1 R10\u003dfp\n      11: (65) if r1 s\u003e 0x1 goto pc+2\n      R0\u003dmap_value(ks\u003d8,vs\u003d8,id\u003d0),min_value\u003d0,max_value\u003d0 R1\u003dinv,min_value\u003d0,max_value\u003d1\n      R2\u003dimm-1,max_value\u003d18446744073709551615,min_align\u003d1 R10\u003dfp\n      12: (0f) r0 +\u003d r1\n      13: (72) *(u8 *)(r0 +0) \u003d 0\n      R0\u003dmap_value_adj(ks\u003d8,vs\u003d8,id\u003d0),min_value\u003d0,max_value\u003d1 R1\u003dinv,min_value\u003d0,max_value\u003d1\n      R2\u003dimm-1,max_value\u003d18446744073709551615,min_align\u003d1 R10\u003dfp\n      14: (b7) r0 \u003d 0\n      15: (95) exit\n\n    What happens is that in the first part ...\n\n       8: (79) r1 \u003d *(u64 *)(r10 -16)\n       9: (b7) r2 \u003d -1\n      10: (2d) if r1 \u003e r2 goto pc+3\n\n    ... r1 carries an unsigned value, and is compared as unsigned\n    against a register carrying an immediate. Verifier deduces in\n    reg_set_min_max() that since the compare is unsigned and operation\n    is greater than (\u003e), that in the fall-through/false case, r1\u0027s\n    minimum bound must be 0 and maximum bound must be r2. Latter is\n    larger than the bound and thus max value is reset back to being\n    \u0027invalid\u0027 aka BPF_REGISTER_MAX_RANGE. Thus, r1 state is now\n    \u0027R1\u003dinv,min_value\u003d0\u0027. The subsequent test ...\n\n      11: (65) if r1 s\u003e 0x1 goto pc+2\n\n    ... is a signed compare of r1 with immediate value 1. Here,\n    verifier deduces in reg_set_min_max() that since the compare\n    is signed this time and operation is greater than (\u003e), that\n    in the fall-through/false case, we can deduce that r1\u0027s maximum\n    bound must be 1, meaning with prior test, we result in r1 having\n    the following state: R1\u003dinv,min_value\u003d0,max_value\u003d1. Given that\n    the actual value this holds is -8, the bounds are wrongly deduced.\n    When this is being added to r0 which holds the map_value(_adj)\n    type, then subsequent store access in above case will go through\n    check_mem_access() which invokes check_map_access_adj(), that\n    will then probe whether the map memory is in bounds based\n    on the min_value and max_value as well as access size since\n    the actual unknown value is min_value \u003c\u003d x \u003c\u003d max_value; commit\n    fce366a9dd0d (\"bpf, verifier: fix alu ops against map_value{,\n    _adj} register types\") provides some more explanation on the\n    semantics.\n\n    It\u0027s worth to note in this context that in the current code,\n    min_value and max_value tracking are used for two things, i)\n    dynamic map value access via check_map_access_adj() and since\n    commit 06c1c049721a (\"bpf: allow helpers access to variable memory\")\n    ii) also enforced at check_helper_mem_access() when passing a\n    memory address (pointer to packet, map value, stack) and length\n    pair to a helper and the length in this case is an unknown value\n    defining an access range through min_value/max_value in that\n    case. The min_value/max_value tracking is /not/ used in the\n    direct packet access case to track ranges. However, the issue\n    also affects case ii), for example, the following crafted program\n    based on the same principle must be rejected as well:\n\n       0: (b7) r2 \u003d 0\n       1: (bf) r3 \u003d r10\n       2: (07) r3 +\u003d -512\n       3: (7a) *(u64 *)(r10 -16) \u003d -8\n       4: (79) r4 \u003d *(u64 *)(r10 -16)\n       5: (b7) r6 \u003d -1\n       6: (2d) if r4 \u003e r6 goto pc+5\n      R1\u003dctx R2\u003dimm0,min_value\u003d0,max_value\u003d0,min_align\u003d2147483648 R3\u003dfp-512\n      R4\u003dinv,min_value\u003d0 R6\u003dimm-1,max_value\u003d18446744073709551615,min_align\u003d1 R10\u003dfp\n       7: (65) if r4 s\u003e 0x1 goto pc+4\n      R1\u003dctx R2\u003dimm0,min_value\u003d0,max_value\u003d0,min_align\u003d2147483648 R3\u003dfp-512\n      R4\u003dinv,min_value\u003d0,max_value\u003d1 R6\u003dimm-1,max_value\u003d18446744073709551615,min_align\u003d1\n      R10\u003dfp\n       8: (07) r4 +\u003d 1\n       9: (b7) r5 \u003d 0\n      10: (6a) *(u16 *)(r10 -512) \u003d 0\n      11: (85) call bpf_skb_load_bytes#26\n      12: (b7) r0 \u003d 0\n      13: (95) exit\n\n    Meaning, while we initialize the max_value stack slot that the\n    verifier thinks we access in the [1,2] range, in reality we\n    pass -7 as length which is interpreted as u32 in the helper.\n    Thus, this issue is relevant also for the case of helper ranges.\n    Resetting both bounds in check_reg_overflow() in case only one\n    of them exceeds limits is also not enough as similar test can be\n    created that uses values which are within range, thus also here\n    learned min value in r1 is incorrect when mixed with later signed\n    test to create a range:\n\n       0: (7a) *(u64 *)(r10 -8) \u003d 0\n       1: (bf) r2 \u003d r10\n       2: (07) r2 +\u003d -8\n       3: (18) r1 \u003d 0xffff880ad081fa00\n       5: (85) call bpf_map_lookup_elem#1\n       6: (15) if r0 \u003d\u003d 0x0 goto pc+7\n      R0\u003dmap_value(ks\u003d8,vs\u003d8,id\u003d0),min_value\u003d0,max_value\u003d0 R10\u003dfp\n       7: (7a) *(u64 *)(r10 -16) \u003d -8\n       8: (79) r1 \u003d *(u64 *)(r10 -16)\n       9: (b7) r2 \u003d 2\n      10: (3d) if r2 \u003e\u003d r1 goto pc+3\n      R0\u003dmap_value(ks\u003d8,vs\u003d8,id\u003d0),min_value\u003d0,max_value\u003d0 R1\u003dinv,min_value\u003d3\n      R2\u003dimm2,min_value\u003d2,max_value\u003d2,min_align\u003d2 R10\u003dfp\n      11: (65) if r1 s\u003e 0x4 goto pc+2\n      R0\u003dmap_value(ks\u003d8,vs\u003d8,id\u003d0),min_value\u003d0,max_value\u003d0\n      R1\u003dinv,min_value\u003d3,max_value\u003d4 R2\u003dimm2,min_value\u003d2,max_value\u003d2,min_align\u003d2 R10\u003dfp\n      12: (0f) r0 +\u003d r1\n      13: (72) *(u8 *)(r0 +0) \u003d 0\n      R0\u003dmap_value_adj(ks\u003d8,vs\u003d8,id\u003d0),min_value\u003d3,max_value\u003d4\n      R1\u003dinv,min_value\u003d3,max_value\u003d4 R2\u003dimm2,min_value\u003d2,max_value\u003d2,min_align\u003d2 R10\u003dfp\n      14: (b7) r0 \u003d 0\n      15: (95) exit\n\n    This leaves us with two options for fixing this: i) to invalidate\n    all prior learned information once we switch signed context, ii)\n    to track min/max signed and unsigned boundaries separately as\n    done in [0]. (Given latter introduces major changes throughout\n    the whole verifier, it\u0027s rather net-next material, thus this\n    patch follows option i), meaning we can derive bounds either\n    from only signed tests or only unsigned tests.) There is still the\n    case of adjust_reg_min_max_vals(), where we adjust bounds on ALU\n    operations, meaning programs like the following where boundaries\n    on the reg get mixed in context later on when bounds are merged\n    on the dst reg must get rejected, too:\n\n       0: (7a) *(u64 *)(r10 -8) \u003d 0\n       1: (bf) r2 \u003d r10\n       2: (07) r2 +\u003d -8\n       3: (18) r1 \u003d 0xffff89b2bf87ce00\n       5: (85) call bpf_map_lookup_elem#1\n       6: (15) if r0 \u003d\u003d 0x0 goto pc+6\n      R0\u003dmap_value(ks\u003d8,vs\u003d8,id\u003d0),min_value\u003d0,max_value\u003d0 R10\u003dfp\n       7: (7a) *(u64 *)(r10 -16) \u003d -8\n       8: (79) r1 \u003d *(u64 *)(r10 -16)\n       9: (b7) r2 \u003d 2\n      10: (3d) if r2 \u003e\u003d r1 goto pc+2\n      R0\u003dmap_value(ks\u003d8,vs\u003d8,id\u003d0),min_value\u003d0,max_value\u003d0 R1\u003dinv,min_value\u003d3\n      R2\u003dimm2,min_value\u003d2,max_value\u003d2,min_align\u003d2 R10\u003dfp\n      11: (b7) r7 \u003d 1\n      12: (65) if r7 s\u003e 0x0 goto pc+2\n      R0\u003dmap_value(ks\u003d8,vs\u003d8,id\u003d0),min_value\u003d0,max_value\u003d0 R1\u003dinv,min_value\u003d3\n      R2\u003dimm2,min_value\u003d2,max_value\u003d2,min_align\u003d2 R7\u003dimm1,max_value\u003d0 R10\u003dfp\n      13: (b7) r0 \u003d 0\n      14: (95) exit\n\n      from 12 to 15: R0\u003dmap_value(ks\u003d8,vs\u003d8,id\u003d0),min_value\u003d0,max_value\u003d0\n      R1\u003dinv,min_value\u003d3 R2\u003dimm2,min_value\u003d2,max_value\u003d2,min_align\u003d2 R7\u003dimm1,min_value\u003d1 R10\u003dfp\n      15: (0f) r7 +\u003d r1\n      16: (65) if r7 s\u003e 0x4 goto pc+2\n      R0\u003dmap_value(ks\u003d8,vs\u003d8,id\u003d0),min_value\u003d0,max_value\u003d0 R1\u003dinv,min_value\u003d3\n      R2\u003dimm2,min_value\u003d2,max_value\u003d2,min_align\u003d2 R7\u003dinv,min_value\u003d4,max_value\u003d4 R10\u003dfp\n      17: (0f) r0 +\u003d r7\n      18: (72) *(u8 *)(r0 +0) \u003d 0\n      R0\u003dmap_value_adj(ks\u003d8,vs\u003d8,id\u003d0),min_value\u003d4,max_value\u003d4 R1\u003dinv,min_value\u003d3\n      R2\u003dimm2,min_value\u003d2,max_value\u003d2,min_align\u003d2 R7\u003dinv,min_value\u003d4,max_value\u003d4 R10\u003dfp\n      19: (b7) r0 \u003d 0\n      20: (95) exit\n\n    Meaning, in adjust_reg_min_max_vals() we must also reset range\n    values on the dst when src/dst registers have mixed signed/\n    unsigned derived min/max value bounds with one unbounded value\n    as otherwise they can be added together deducing false boundaries.\n    Once both boundaries are established from either ALU ops or\n    compare operations w/o mixing signed/unsigned insns, then they\n    can safely be added to other regs also having both boundaries\n    established. Adding regs with one unbounded side to a map value\n    where the bounded side has been learned w/o mixing ops is\n    possible, but the resulting map value won\u0027t recover from that,\n    meaning such op is considered invalid on the time of actual\n    access. Invalid bounds are set on the dst reg in case i) src reg,\n    or ii) in case dst reg already had them. The only way to recover\n    would be to perform i) ALU ops but only \u0027add\u0027 is allowed on map\n    value types or ii) comparisons, but these are disallowed on\n    pointers in case they span a range. This is fine as only BPF_JEQ\n    and BPF_JNE may be performed on PTR_TO_MAP_VALUE_OR_NULL registers\n    which potentially turn them into PTR_TO_MAP_VALUE type depending\n    on the branch, so only here min/max value cannot be invalidated\n    for them.\n\n    In terms of state pruning, value_from_signed is considered\n    as well in states_equal() when dealing with adjusted map values.\n    With regards to breaking existing programs, there is a small\n    risk, but use-cases are rather quite narrow where this could\n    occur and mixing compares probably unlikely.\n\n    Joint work with Josef and Edward.\n\n      [0] https://lists.iovisor.org/pipermail/iovisor-dev/2017-June/000822.html\n\n    Fixes: 484611357c19 (\"bpf: allow access into map value arrays\")\n    Reported-by: Edward Cree \u003cecree@solarflare.com\u003e\n    Signed-off-by: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Signed-off-by: Edward Cree \u003cecree@solarflare.com\u003e\n    Signed-off-by: Josef Bacik \u003cjbacik@fb.com\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Greg Kroah-Hartman \u003cgregkh@linuxfoundation.org\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 155147252d31bdd6cd2b88d534b6e09c6af18e62\nAuthor: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\nDate:   Fri Mar 31 02:24:02 2017 +0200\n\n    bpf, verifier: fix alu ops against map_value{, _adj} register types\n\n    [ Upstream commit fce366a9dd0ddc47e7ce05611c266e8574a45116 ]\n\n    While looking into map_value_adj, I noticed that alu operations\n    directly on the map_value() resp. map_value_adj() register (any\n    alu operation on a map_value() register will turn it into a\n    map_value_adj() typed register) are not sufficiently protected\n    against some of the operations. Two non-exhaustive examples are\n    provided that the verifier needs to reject:\n\n     i) BPF_AND on r0 (map_value_adj):\n\n      0: (bf) r2 \u003d r10\n      1: (07) r2 +\u003d -8\n      2: (7a) *(u64 *)(r2 +0) \u003d 0\n      3: (18) r1 \u003d 0xbf842a00\n      5: (85) call bpf_map_lookup_elem#1\n      6: (15) if r0 \u003d\u003d 0x0 goto pc+2\n       R0\u003dmap_value(ks\u003d8,vs\u003d48,id\u003d0),min_value\u003d0,max_value\u003d0 R10\u003dfp\n      7: (57) r0 \u0026\u003d 8\n      8: (7a) *(u64 *)(r0 +0) \u003d 22\n       R0\u003dmap_value_adj(ks\u003d8,vs\u003d48,id\u003d0),min_value\u003d0,max_value\u003d8 R10\u003dfp\n      9: (95) exit\n\n      from 6 to 9: R0\u003dinv,min_value\u003d0,max_value\u003d0 R10\u003dfp\n      9: (95) exit\n      processed 10 insns\n\n    ii) BPF_ADD in 32 bit mode on r0 (map_value_adj):\n\n      0: (bf) r2 \u003d r10\n      1: (07) r2 +\u003d -8\n      2: (7a) *(u64 *)(r2 +0) \u003d 0\n      3: (18) r1 \u003d 0xc24eee00\n      5: (85) call bpf_map_lookup_elem#1\n      6: (15) if r0 \u003d\u003d 0x0 goto pc+2\n       R0\u003dmap_value(ks\u003d8,vs\u003d48,id\u003d0),min_value\u003d0,max_value\u003d0 R10\u003dfp\n      7: (04) (u32) r0 +\u003d (u32) 0\n      8: (7a) *(u64 *)(r0 +0) \u003d 22\n       R0\u003dmap_value_adj(ks\u003d8,vs\u003d48,id\u003d0),min_value\u003d0,max_value\u003d0 R10\u003dfp\n      9: (95) exit\n\n      from 6 to 9: R0\u003dinv,min_value\u003d0,max_value\u003d0 R10\u003dfp\n      9: (95) exit\n      processed 10 insns\n\n    Issue is, while min_value / max_value boundaries for the access\n    are adjusted appropriately, we change the pointer value in a way\n    that cannot be sufficiently tracked anymore from its origin.\n    Operations like BPF_{AND,OR,DIV,MUL,etc} on a destination register\n    that is PTR_TO_MAP_VALUE{,_ADJ} was probably unintended, in fact,\n    all the test cases coming with 484611357c19 (\"bpf: allow access\n    into map value arrays\") perform BPF_ADD only on the destination\n    register that is PTR_TO_MAP_VALUE_ADJ.\n\n    Only for UNKNOWN_VALUE register types such operations make sense,\n    f.e. with unknown memory content fetched initially from a constant\n    offset from the map value memory into a register. That register is\n    then later tested against lower / upper bounds, so that the verifier\n    can then do the tracking of min_value / max_value, and properly\n    check once that UNKNOWN_VALUE register is added to the destination\n    register with type PTR_TO_MAP_VALUE{,_ADJ}. This is also what the\n    original use-case is solving. Note, tracking on what is being\n    added is done through adjust_reg_min_max_vals() and later access\n    to the map value enforced with these boundaries and the given offset\n    from the insn through check_map_access_adj().\n\n    Tests will fail for non-root environment due to prohibited pointer\n    arithmetic, in particular in check_alu_op(), we bail out on the\n    is_pointer_value() check on the dst_reg (which is false in root\n    case as we allow for pointer arithmetic via env-\u003eallow_ptr_leaks).\n\n    Similarly to PTR_TO_PACKET, one way to fix it is to restrict the\n    allowed operations on PTR_TO_MAP_VALUE{,_ADJ} registers to 64 bit\n    mode BPF_ADD. The test_verifier suite runs fine after the patch\n    and it also rejects mentioned test cases.\n\n    Fixes: 484611357c19 (\"bpf: allow access into map value arrays\")\n    Signed-off-by: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Reviewed-by: Josef Bacik \u003cjbacik@fb.com\u003e\n    Acked-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Greg Kroah-Hartman \u003cgregkh@linuxfoundation.org\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 4706daec7b6714a59f210cdfbc1c899e8d5c7307\nAuthor: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\nDate:   Thu May 18 03:00:06 2017 +0200\n\n    bpf: adjust verifier heuristics\n\n    [ Upstream commit 3c2ce60bdd3d57051bf85615deec04a694473840 ]\n\n    Current limits with regards to processing program paths do not\n    really reflect today\u0027s needs anymore due to programs becoming\n    more complex and verifier smarter, keeping track of more data\n    such as const ALU operations, alignment tracking, spilling of\n    PTR_TO_MAP_VALUE_ADJ registers, and other features allowing for\n    smarter matching of what LLVM generates.\n\n    This also comes with the side-effect that we result in fewer\n    opportunities to prune search states and thus often need to do\n    more work to prove safety than in the past due to different\n    register states and stack layout where we mismatch. Generally,\n    it\u0027s quite hard to determine what caused a sudden increase in\n    complexity, it could be caused by something as trivial as a\n    single branch somewhere at the beginning of the program where\n    LLVM assigned a stack slot that is marked differently throughout\n    other branches and thus causing a mismatch, where verifier\n    then needs to prove safety for the whole rest of the program.\n    Subsequently, programs with even less than half the insn size\n    limit can get rejected. We noticed that while some programs\n    load fine under pre 4.11, they get rejected due to hitting\n    limits on more recent kernels. We saw that in the vast majority\n    of cases (90+%) pruning failed due to register mismatches. In\n    case of stack mismatches, majority of cases failed due to\n    different stack slot types (invalid, spill, misc) rather than\n    differences in spilled registers.\n\n    This patch makes pruning more aggressive by also adding markers\n    that sit at conditional jumps as well. Currently, we only mark\n    jump targets for pruning. For example in direct packet access,\n    these are usually error paths where we bail out. We found that\n    adding these markers, it can reduce number of processed insns\n    by up to 30%. Another option is to ignore reg-\u003eid in probing\n    PTR_TO_MAP_VALUE_OR_NULL registers, which can help pruning\n    slightly as well by up to 7% observed complexity reduction as\n    stand-alone. Meaning, if a previous path with register type\n    PTR_TO_MAP_VALUE_OR_NULL for map X was found to be safe, then\n    in the current state a PTR_TO_MAP_VALUE_OR_NULL register for\n    the same map X must be safe as well. Last but not least the\n    patch also adds a scheduling point and bumps the current limit\n    for instructions to be processed to a more adequate value.\n\n    Signed-off-by: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Acked-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Greg Kroah-Hartman \u003cgregkh@linuxfoundation.org\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit ccbff7ed5a69d67b88cdd5802628cdaffcf201eb\nAuthor: John Fastabend \u003cjohn.fastabend@gmail.com\u003e\nDate:   Sun Jul 2 02:13:30 2017 +0200\n\n    bpf, verifier: add additional patterns to evaluate_reg_imm_alu\n\n    [ Upstream commit 43188702b3d98d2792969a3377a30957f05695e6 ]\n\n    Currently the verifier does not track imm across alu operations when\n    the source register is of unknown type. This adds additional pattern\n    matching to catch this and track imm. We\u0027ve seen LLVM generating this\n    pattern while working on cilium.\n\n    Signed-off-by: John Fastabend \u003cjohn.fastabend@gmail.com\u003e\n    Acked-by: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Acked-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Greg Kroah-Hartman \u003cgregkh@linuxfoundation.org\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit a8a14af3f51bd3aaedebb89dd8c5293c653f4a69\nAuthor: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\nDate:   Thu Jun 29 03:04:59 2017 +0200\n\n    bpf: prevent leaking pointer via xadd on unpriviledged\n\n    commit 6bdf6abc56b53103324dfd270a86580306e1a232 upstream.\n\n    Leaking kernel addresses on unpriviledged is generally disallowed,\n    for example, verifier rejects the following:\n\n      0: (b7) r0 \u003d 0\n      1: (18) r2 \u003d 0xffff897e82304400\n      3: (7b) *(u64 *)(r1 +48) \u003d r2\n      R2 leaks addr into ctx\n\n    Doing pointer arithmetic on them is also forbidden, so that they\n    don\u0027t turn into unknown value and then get leaked out. However,\n    there\u0027s xadd as a special case, where we don\u0027t check the src reg\n    for being a pointer register, e.g. the following will pass:\n\n      0: (b7) r0 \u003d 0\n      1: (7b) *(u64 *)(r1 +48) \u003d r0\n      2: (18) r2 \u003d 0xffff897e82304400 ; map\n      4: (db) lock *(u64 *)(r1 +48) +\u003d r2\n      5: (95) exit\n\n    We could store the pointer into skb-\u003ecb, loose the type context,\n    and then read it out from there again to leak it eventually out\n    of a map value. Or more easily in a different variant, too:\n\n       0: (bf) r6 \u003d r1\n       1: (7a) *(u64 *)(r10 -8) \u003d 0\n       2: (bf) r2 \u003d r10\n       3: (07) r2 +\u003d -8\n       4: (18) r1 \u003d 0x0\n       6: (85) call bpf_map_lookup_elem#1\n       7: (15) if r0 \u003d\u003d 0x0 goto pc+3\n       R0\u003dmap_value(ks\u003d8,vs\u003d8,id\u003d0),min_value\u003d0,max_value\u003d0 R6\u003dctx R10\u003dfp\n       8: (b7) r3 \u003d 0\n       9: (7b) *(u64 *)(r0 +0) \u003d r3\n      10: (db) lock *(u64 *)(r0 +0) +\u003d r6\n      11: (b7) r0 \u003d 0\n      12: (95) exit\n\n      from 7 to 11: R0\u003dinv,min_value\u003d0,max_value\u003d0 R6\u003dctx R10\u003dfp\n      11: (b7) r0 \u003d 0\n      12: (95) exit\n\n    Prevent this by checking xadd src reg for pointer types. Also\n    add a couple of test cases related to this.\n\n    Fixes: 1be7f75d1668 (\"bpf: enable non-root eBPF programs\")\n    Fixes: 17a5267067f3 (\"bpf: verifier (add verifier core)\")\n    Signed-off-by: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Acked-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Acked-by: Martin KaFai Lau \u003ckafai@fb.com\u003e\n    Acked-by: Edward Cree \u003cecree@solarflare.com\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Greg Kroah-Hartman \u003cgregkh@linuxfoundation.org\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit a2a655b8d49d19263b4779d41600bc4f470aa1be\nAuthor: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\nDate:   Wed Jan 18 15:14:17 2017 +0100\n\n    bpf: don\u0027t trigger OOM killer under pressure with map alloc\n\n    [ Upstream commit d407bd25a204bd66b7346dde24bd3d37ef0e0b05 ]\n\n    This patch adds two helpers, bpf_map_area_alloc() and bpf_map_area_free(),\n    that are to be used for map allocations. Using kmalloc() for very large\n    allocations can cause excessive work within the page allocator, so i) fall\n    back earlier to vmalloc() when the attempt is considered costly anyway,\n    and even more importantly ii) don\u0027t trigger OOM killer with any of the\n    allocators.\n\n    Since this is based on a user space request, for example, when creating\n    maps with element pre-allocation, we really want such requests to fail\n    instead of killing other user space processes.\n\n    Also, don\u0027t spam the kernel log with warnings should any of the allocations\n    fail under pressure. Given that, we can make backend selection in\n    bpf_map_area_alloc() generic, and convert all maps over to use this API\n    for spots with potentially large allocation requests.\n\n    Note, replacing the one kmalloc_array() is fine as overflow checks happen\n    earlier in htab_map_alloc(), since it must also protect the multiplication\n    for vmalloc() should kmalloc_array() fail.\n\n    Signed-off-by: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Acked-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Sasha Levin \u003calexander.levin@verizon.com\u003e\n    Signed-off-by: Greg Kroah-Hartman \u003cgregkh@linuxfoundation.org\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit bfda5f80c50db00b4d2fe4e236e8381768599768\nAuthor: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\nDate:   Tue Jun 6 18:38:04 2017 +0200\n\n    FROMLIST: bpf: cgroup skb progs cannot access ld_abs/ind\n\n    Commit fb9a307d11d6 (\"bpf: Allow CGROUP_SKB eBPF program to\n    access sk_buff\") enabled programs of BPF_PROG_TYPE_CGROUP_SKB\n    type to use ld_abs/ind instructions. However, at this point,\n    we cannot use them, since offsets relative to SKF_LL_OFF will\n    end up pointing skb_mac_header(skb) out of bounds since in the\n    egress path it is not yet set at that point in time, but only\n    after __dev_queue_xmit() did a general reset on the mac header.\n    bpf_internal_load_pointer_neg_helper() will then end up reading\n    data from a wrong offset.\n\n    BPF_PROG_TYPE_CGROUP_SKB programs can use bpf_skb_load_bytes()\n    already to access packet data, which is also more flexible than\n    the insns carried over from cBPF.\n\n    Fixes: fb9a307d11d6 (\"bpf: Allow CGROUP_SKB eBPF program to access sk_buff\")\n    Signed-off-by: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Acked-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Cc: Chenbo Feng \u003cfengc@google.com\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n\n    (url: http://patchwork.ozlabs.org/patch/771946/)\n\n    Signed-off-by: Chenbo Feng \u003cfengc@google.com\u003e\n    Bug: 30950746\n    Change-Id: Ia32ac79d8c0d18f811ec101897284a8b60cb042a\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit f7021caa40d6683b832385c7de022f6742124749\nAuthor: Chenbo Feng \u003cfengc@google.com\u003e\nDate:   Fri Jun 2 17:24:31 2017 -0700\n\n    FROMLIST: [net-next,v2,2/2] bpf: Remove the capability check for cgroup skb eBPF program\n\n    Currently loading a cgroup skb eBPF program require a CAP_SYS_ADMIN\n    capability while attaching the program to a cgroup only requires the\n    user have CAP_NET_ADMIN privilege. We can escape the capability\n    check when load the program just like socket filter program to make\n    the capability requirement consistent.\n\n    Change since v1:\n    Change the code style in order to be compliant with checkpatch.pl\n    preference\n\n    (url: http://patchwork.ozlabs.org/patch/769460/)\n    Signed-off-by: Chenbo Feng \u003cfengc@google.com\u003e\n    Bug: 30950746\n    Change-Id: Ibe51235127d6f9349b8f563ad31effc061b278ed\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit d317d86203c7bd0023125fedf7a695e0713c7a69\nAuthor: Chenbo Feng \u003cfengc@google.com\u003e\nDate:   Fri Jun 2 17:04:59 2017 -0700\n\n    FROMLIST: [net-next,v2,1/2] bpf: Allow CGROUP_SKB eBPF program to access sk_buff\n\n    This allows cgroup eBPF program to classify packet based on their\n    protocol or other detail information. Currently program need\n    CAP_NET_ADMIN privilege to attach a cgroup eBPF program, and A\n    process with CAP_NET_ADMIN can already see all packets on the system,\n    for example, by creating an iptables rules that causes the packet to\n    be passed to userspace via NFLOG.\n\n    (url: http://patchwork.ozlabs.org/patch/769459/)\n\n    Signed-off-by: Chenbo Feng \u003cfengc@google.com\u003e\n    Bug: 30950746\n    Change-Id: I11bef84ce26cf8b8f1b89483c32a7fcdd61ae926\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 5c5b00bb89839f634e7ea6561c04376458567709\nAuthor: Daniel Mack \u003cdaniel@zonque.org\u003e\nDate:   Mon Nov 28 14:11:04 2016 +0100\n\n    UPSTREAM: bpf: cgroup: fix documentation of __cgroup_bpf_update()\n\n    There\u0027s a \u0027not\u0027 missing in one paragraph. Add it.\n\n    Fixes: 3007098494be (\"cgroup: add support for eBPF programs\")\n    Signed-off-by: Daniel Mack \u003cdaniel@zonque.org\u003e\n    Reported-by: Rami Rosen \u003croszenrami@gmail.com\u003e\n    Acked-by: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Acked-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n\n    Fixes: Change-Id: I3df35d8d3b1261503f9b5bcd90b18c9358f1ac28\n           (\"cgroup: add support for eBPF programs\")\n    (cherry picked from commit 01ae87eab53675cbdabd5c4d727c4a35e397cce0)\n    Signed-off-by: Amit Pundir \u003camit.pundir@linaro.org\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 8f630d15e6f6bbf94358b332177fcda77ba85c7f\nAuthor: Alexei Starovoitov \u003cast@fb.com\u003e\nDate:   Fri Feb 10 20:28:24 2017 -0800\n\n    BACKPORT: bpf: introduce BPF_F_ALLOW_OVERRIDE flag\n\n    If BPF_F_ALLOW_OVERRIDE flag is used in BPF_PROG_ATTACH command\n    to the given cgroup the descendent cgroup will be able to override\n    effective bpf program that was inherited from this cgroup.\n    By default it\u0027s not passed, therefore override is disallowed.\n\n    Examples:\n    1.\n    prog X attached to /A with default\n    prog Y fails to attach to /A/B and /A/B/C\n    Everything under /A runs prog X\n\n    2.\n    prog X attached to /A with allow_override.\n    prog Y fails to attach to /A/B with default (non-override)\n    prog M attached to /A/B with allow_override.\n    Everything under /A/B runs prog M only.\n\n    3.\n    prog X attached to /A with allow_override.\n    prog Y fails to attach to /A with default.\n    The user has to detach first to switch the mode.\n\n    In the future this behavior may be extended with a chain of\n    non-overridable programs.\n\n    Also fix the bug where detach from cgroup where nothing is attached\n    was not throwing error. Return ENOENT in such case.\n\n    Add several testcases and adjust libbpf.\n\n    Fixes: 3007098494be (\"cgroup: add support for eBPF programs\")\n    Signed-off-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Acked-by: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Acked-by: Tejun Heo \u003ctj@kernel.org\u003e\n    Acked-by: Daniel Mack \u003cdaniel@zonque.org\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n\n    Fixes: Change-Id: I3df35d8d3b1261503f9b5bcd90b18c9358f1ac28\n           (\"cgroup: add support for eBPF programs\")\n    [AmitP: Refactored original patch for android-4.9 where libbpf sources\n            are in samples/bpf/ and test_cgrp2_attach2, test_cgrp2_sock,\n            and test_cgrp2_sock2 sample tests do not exist.]\n    (cherry picked from commit 7f677633379b4abb3281cdbe7e7006f049305c03)\n    Signed-off-by: Amit Pundir \u003camit.pundir@linaro.org\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 0e3a272e683c012bfa1367a4faa8a6fecdce731a\nAuthor: Daniel Mack \u003cdaniel@zonque.org\u003e\nDate:   Wed Nov 23 16:52:30 2016 +0100\n\n    UPSTREAM: samples: bpf: add userspace example for attaching eBPF programs to cgroups\n\n    Cherry-pick from commit d8c5b17f2bc0de09fbbfa14d90e8168163a579e7\n\n    Add a simple userpace program to demonstrate the new API to attach eBPF\n    programs to cgroups. This is what it does:\n\n     * Create arraymap in kernel with 4 byte keys and 8 byte values\n\n     * Load eBPF program\n\n       The eBPF program accesses the map passed in to store two pieces of\n       information. The number of invocations of the program, which maps\n       to the number of packets received, is stored to key 0. Key 1 is\n       incremented on each iteration by the number of bytes stored in\n       the skb.\n\n     * Detach any eBPF program previously attached to the cgroup\n\n     * Attach the new program to the cgroup using BPF_PROG_ATTACH\n\n     * Once a second, read map[0] and map[1] to see how many bytes and\n       packets were seen on any socket of tasks in the given cgroup.\n\n    The program takes a cgroup path as 1st argument, and either \"ingress\"\n    or \"egress\" as 2nd. Optionally, \"drop\" can be passed as 3rd argument,\n    which will make the generated eBPF program return 0 instead of 1, so\n    the kernel will drop the packet.\n\n    libbpf gained two new wrappers for the new syscall commands.\n\n    Signed-off-by: Daniel Mack \u003cdaniel@zonque.org\u003e\n    Acked-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Bug: 30950746\n    Change-Id: I011436a755abd62050edd22e47995c166a0bd8a2\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit bee81948a552ad018cacc5661f10eac0c6360088\nAuthor: Daniel Mack \u003cdaniel@zonque.org\u003e\nDate:   Wed Nov 23 16:52:25 2016 +0100\n\n    UPSTREAM: bpf: add new prog type for cgroup socket filtering\n\n    Cherry-pick from commit 0e33661de493db325435d565a4a722120ae4cbf3\n\n    This program type is similar to BPF_PROG_TYPE_SOCKET_FILTER, except that\n    it does not allow BPF_LD_[ABS|IND] instructions and hooks up the\n    bpf_skb_load_bytes() helper.\n\n    Programs of this type will be attached to cgroups for network filtering\n    and accounting.\n\n    Signed-off-by: Daniel Mack \u003cdaniel@zonque.org\u003e\n    Acked-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Bug: 30950746\n    Change-Id: I7b9e063d5d7a91da80917c6d353a60b877133752\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit a17f4d37e594e967e30573d58bc07141b3d72dde\nAuthor: Willem de Bruijn \u003cwillemb@google.com\u003e\nDate:   Tue Apr 11 14:08:08 2017 -0400\n\n    BACKPORT: UPSTREAM: bpf: pass sk to helper functions\n\n    Cherrypick from commit 8f917bba0042f1e3b7693743fbe9782709e936e7\n\n    BPF helper functions access socket fields through skb-\u003esk. This is not\n    set in ingress cgroup and socket filters. The association is only made\n    in skb_set_owner_r once the filter has accepted the packet. Sk is\n    available as socket lookup has taken place.\n\n    Temporarily set skb-\u003esk to sk in these cases.\n\n    Signed-off-by: Willem de Bruijn \u003cwillemb@google.com\u003e\n    Acked-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Bug: 30950746\n    Change-Id: Ifcbcbe2ab2882dc79c56f9707be1d6aef08c7fd3\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit f8987f871e8401a504d87ef5ef5040cf1a02d16a\nAuthor: Daniel Mack \u003cdaniel@zonque.org\u003e\nDate:   Wed Nov 23 16:52:27 2016 +0100\n\n    UPSTREAM: bpf: add BPF_PROG_ATTACH and BPF_PROG_DETACH commands\n\n    Cherry-pick from commit f4324551489e8781d838f941b7aee4208e52e8bf\n\n    Extend the bpf(2) syscall by two new commands, BPF_PROG_ATTACH and\n    BPF_PROG_DETACH which allow attaching and detaching eBPF programs\n    to a target.\n\n    On the API level, the target could be anything that has an fd in\n    userspace, hence the name of the field in union bpf_attr is called\n    \u0027target_fd\u0027.\n\n    When called with BPF_ATTACH_TYPE_CGROUP_INET_{E,IN}GRESS, the target is\n    expected to be a valid file descriptor of a cgroup v2 directory which\n    has the bpf controller enabled. These are the only use-cases\n    implemented by this patch at this point, but more can be added.\n\n    If a program of the given type already exists in the given cgroup,\n    the program is swapped automically, so userspace does not have to drop\n    an existing program first before installing a new one, which would\n    otherwise leave a gap in which no program is attached.\n\n    For more information on the propagation logic to subcgroups, please\n    refer to the bpf cgroup controller implementation.\n\n    The API is guarded by CAP_NET_ADMIN.\n\n    Signed-off-by: Daniel Mack \u003cdaniel@zonque.org\u003e\n    Acked-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Bug: 30950746\n    Change-Id: Iab156859332166835d51e1e6f64e5cb8b81870f2\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 1b3d5978e1866f150ccad2ff421faab89589e5f5\nAuthor: Tejun Heo \u003ctj@kernel.org\u003e\nDate:   Fri Nov 20 15:55:52 2015 -0500\n\n    kernfs: implement kernfs_walk_and_get()\n\n    Implement kernfs_walk_and_get() which is similar to\n    kernfs_find_and_get() but can walk a path instead of just a name.\n\n    v2: Use strlcpy() instead of strlen() + memcpy() as suggested by\n        David.\n\n    Signed-off-by: Tejun Heo \u003ctj@kernel.org\u003e\n    Acked-by: Greg Kroah-Hartman \u003cgregkh@linuxfoundation.org\u003e\n    Cc: David Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 44f016625b070038fb36eee2bc346238683a1022\nAuthor: Johannes Weiner \u003channes@cmpxchg.org\u003e\nDate:   Thu Jan 31 10:41:54 2019 +1100\n\n    BACKPORT: fs: kernfs: add poll file operation\n\n    Patch series \"psi: pressure stall monitors\", v3.\n\n    Android is adopting psi to detect and remedy memory pressure that results\n    in stuttering and decreased responsiveness on mobile devices.\n\n    Psi gives us the stall information, but because we\u0027re dealing with\n    latencies in the millisecond range, periodically reading the pressure\n    files to detect stalls in a timely fashion is not feasible.  Psi also\n    doesn\u0027t aggregate its averages at a high enough frequency right now.\n\n    This patch series extends the psi interface such that users can configure\n    sensitive latency thresholds and use poll() and friends to be notified\n    when these are breached.\n\n    As high-frequency aggregation is costly, it implements an aggregation\n    method that is optimized for fast, short-interval averaging, and makes the\n    aggregation frequency adaptive, such that high-frequency updates only\n    happen while monitored stall events are actively occurring.\n\n    With these patches applied, Android can monitor for, and ward off,\n    mounting memory shortages before they cause problems for the user.  For\n    example, using memory stall monitors in userspace low memory killer daemon\n    (lmkd) we can detect mounting pressure and kill less important processes\n    before device becomes visibly sluggish.  In our memory stress testing psi\n    memory monitors produce roughly 10x less false positives compared to\n    vmpressure signals.  Having ability to specify multiple triggers for the\n    same psi metric allows other parts of Android framework to monitor memory\n    state of the device and act accordingly.\n\n    The new interface is straightforward.  The user opens one of the pressure\n    files for writing and writes a trigger description into the file\n    descriptor that defines the stall state - some or full, and the maximum\n    stall time over a given window of time.  E.g.:\n\n            /* Signal when stall time exceeds 100ms of a 1s window */\n            char trigger[] \u003d \"full 100000 1000000\";\n            fd \u003d open(\"/proc/pressure/memory\");\n            write(fd, trigger, sizeof(trigger));\n            while (poll() \u003e\u003d 0) {\n                    ...\n            }\n            close(fd);\n\n    When the monitored stall state is entered, psi adapts its aggregation\n    frequency according to what the configured time window requires in order\n    to emit event signals in a timely fashion.  Once the stalling subsides,\n    aggregation reverts back to normal.\n\n    The trigger is associated with the open file descriptor.  To stop\n    monitoring, the user only needs to close the file descriptor and the\n    trigger is discarded.\n\n    Patches 1-4 prepare the psi code for polling support.  Patch 5 implements\n    the adaptive polling logic, the pressure growth detection optimized for\n    short intervals, and hooks up write() and poll() on the pressure files.\n\n    The patches were developed in collaboration with Johannes Weiner.\n\n    This patch (of 5):\n\n    Kernfs has a standardized poll/notification mechanism for waking all\n    pollers on all fds when a filesystem node changes.  To allow polling for\n    custom events, add a .poll callback that can override the default.\n\n    This is in preparation for pollable cgroup pressure files which have\n    per-fd trigger configurations.\n\n    Link: http://lkml.kernel.org/r/20190124211518.244221-2-surenb@google.com\n    Signed-off-by: Johannes Weiner \u003channes@cmpxchg.org\u003e\n    Signed-off-by: Suren Baghdasaryan \u003csurenb@google.com\u003e\n    Cc: Dennis Zhou \u003cdennis@kernel.org\u003e\n    Cc: Ingo Molnar \u003cmingo@redhat.com\u003e\n    Cc: Jens Axboe \u003caxboe@kernel.dk\u003e\n    Cc: Li Zefan \u003clizefan@huawei.com\u003e\n    Cc: Peter Zijlstra \u003cpeterz@infradead.org\u003e\n    Cc: Tejun Heo \u003ctj@kernel.org\u003e\n    Signed-off-by: Andrew Morton \u003cakpm@linux-foundation.org\u003e\n    Signed-off-by: Stephen Rothwell \u003csfr@canb.auug.org.au\u003e\n\n    (cherry picked from commit: 147e1a97c4a0bdd43f55a582a9416bb9092563a9)\n\n    Conflicts:\n            fs/kernfs/file.c\n            include/linux/kernfs.h\n\n    1. replaced __poll_t with unsigned int.\n    2. replaced kernfs_dentry_node() with dentry-\u003ed_fsdata\n    3. replaced EPOLLERR/EPOLLPRI with POLLERR/POLLPRI (values are the same)\n\n    Bug: 127712811\n    Test: lmkd in PSI mode\n    Change-Id: Ic2bed334d05aec62f4e695f263893c3057921c55\n    Signed-off-by: Suren Baghdasaryan \u003csurenb@google.com\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit f86fda4c2673b0798c59ecf4490d5710f76bd3a0\nAuthor: Tejun Heo \u003ctj@kernel.org\u003e\nDate:   Tue Dec 27 14:49:03 2016 -0500\n\n    UPSTREAM: kernfs: add kernfs_ops-\u003eopen/release() callbacks\n\n    Add -\u003eopen/release() methods to kernfs_ops.  -\u003eopen() is called when\n    the file is opened and -\u003erelease() when the file is either released or\n    severed.  These callbacks can be used, for example, to manage\n    persistent caching objects over multiple seq_file iterations.\n\n    Signed-off-by: Tejun Heo \u003ctj@kernel.org\u003e\n    Acked-by: Greg Kroah-Hartman \u003cgregkh@linuxfoundation.org\u003e\n    Acked-by: Acked-by: Zefan Li \u003clizefan@huawei.com\u003e\n\n    (cherry picked from commit 0e67db2f9fe91937e798e3d7d22c50a8438187e1)\n\n    Bug: 111308141\n    Test: modified lmkd to use PSI and tested using lmkd_unit_test\n\n    Change-Id: Id06e9d5c6da1280bcdd4dc86309dcfaf52b8f9a4\n    Signed-off-by: Suren Baghdasaryan \u003csurenb@google.com\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 4ce51c00483e190498157a34dde66876d5aeb395\nAuthor: Aditya Kali \u003cadityakali@google.com\u003e\nDate:   Fri Jan 29 02:54:04 2016 -0600\n\n    kernfs: Add API to generate relative kernfs path\n\n    The new function kernfs_path_from_node() generates and returns kernfs\n    path of a given kernfs_node relative to a given parent kernfs_node.\n\n    Signed-off-by: Aditya Kali \u003cadityakali@google.com\u003e\n    Signed-off-by: Serge E. Hallyn \u003cserge.hallyn@canonical.com\u003e\n    Acked-by: Greg Kroah-Hartman \u003cgregkh@linuxfoundation.org\u003e\n    Signed-off-by: Tejun Heo \u003ctj@kernel.org\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit be868eec9d8f486310ebaecfd42e88089c10bf77\nAuthor: Aditya Kali \u003cadityakali@google.com\u003e\nDate:   Fri Jan 29 02:54:05 2016 -0600\n\n    sched: new clone flag CLONE_NEWCGROUP for cgroup namespace\n\n    CLONE_NEWCGROUP will be used to create new cgroup namespace.\n\n    Signed-off-by: Aditya Kali \u003cadityakali@google.com\u003e\n    Signed-off-by: Serge Hallyn \u003cserge.hallyn@canonical.com\u003e\n    Signed-off-by: Tejun Heo \u003ctj@kernel.org\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 566a2b0fcc5b66893329385bf9207d18ed5a3e97\nAuthor: Daniel Mack \u003cdaniel@zonque.org\u003e\nDate:   Wed Nov 23 16:52:26 2016 +0100\n\n    UPSTREAM: cgroup: add support for eBPF programs\n\n    Cherry-pick from commit 3007098494bec614fb55dee7bc0410bb7db5ad18\n\n    This patch adds two sets of eBPF program pointers to struct cgroup.\n    One for such that are directly pinned to a cgroup, and one for such\n    that are effective for it.\n\n    To illustrate the logic behind that, assume the following example\n    cgroup hierarchy.\n\n      A - B - C\n            \\ D - E\n\n    If only B has a program attached, it will be effective for B, C, D\n    and E. If D then attaches a program itself, that will be effective for\n    both D and E, and the program in B will only affect B and C. Only one\n    program of a given type is effective for a cgroup.\n\n    Attaching and detaching programs will be done through the bpf(2)\n    syscall. For now, ingress and egress inet socket filtering are the\n    only supported use-cases.\n\n    Signed-off-by: Daniel Mack \u003cdaniel@zonque.org\u003e\n    Acked-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Bug: 30950746\n    Change-Id: I3df35d8d3b1261503f9b5bcd90b18c9358f1ac28\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 0216c431c114d91feb93ebda7b55748f4002754c\nAuthor: Tejun Heo \u003ctj@kernel.org\u003e\nDate:   Thu Jan 26 16:47:28 2017 -0500\n\n    cgroup: don\u0027t online subsystems before cgroup_name/path() are operational\n\n    commit 07cd12945551b63ecb1a349d50a6d69d1d6feb4a upstream.\n\n    While refactoring cgroup creation, a5bca2152036 (\"cgroup: factor out\n    cgroup_create() out of cgroup_mkdir()\") incorrectly onlined subsystems\n    before the new cgroup is associated with it kernfs_node.  This is fine\n    for cgroup proper but cgroup_name/path() depend on the associated\n    kernfs_node and if a subsystem makes the new cgroup_subsys_state\n    visible, which they\u0027re allowed to after onlining, it can lead to NULL\n    dereference.\n\n    The current code performs cgroup creation and subsystem onlining in\n    cgroup_create() and cgroup_mkdir() makes the cgroup and subsystems\n    visible afterwards.  There\u0027s no reason to online the subsystems early\n    and we can simply drop cgroup_apply_control_enable() call from\n    cgroup_create() so that the subsystems are onlined and made visible at\n    the same time.\n\n    Signed-off-by: Tejun Heo \u003ctj@kernel.org\u003e\n    Reported-by: Konstantin Khlebnikov \u003ckhlebnikov@yandex-team.ru\u003e\n    Fixes: a5bca2152036 (\"cgroup: factor out cgroup_create() out of cgroup_mkdir()\")\n    Signed-off-by: Greg Kroah-Hartman \u003cgregkh@linuxfoundation.org\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 6246469ce54c6317ea480fd4263753fad8b078f6\nAuthor: Tejun Heo \u003ctj@kernel.org\u003e\nDate:   Thu Sep 29 15:49:40 2016 +0200\n\n    cgroup: fix error handling regressions in proc_cgroup_show() and cgroup_release_agent()\n\n    4c737b41de7f (\"cgroup: make cgroup_path() and friends behave in the\n    style of strlcpy()\") broke error handling in proc_cgroup_show() and\n    cgroup_release_agent() by not handling negative return values from\n    cgroup_path_ns_locked().  Fix it.\n\n    Reported-by: Dan Carpenter \u003cdan.carpenter@oracle.com\u003e\n    Signed-off-by: Tejun Heo \u003ctj@kernel.org\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n    Change-Id: If1dbcdb90f9eefbfc2aa245a8fc4da5b15e23296\n\ncommit 0ccc12c3c66392dfe9c14caeb8a3f8f35a8793ba\nAuthor: Tejun Heo \u003ctj@kernel.org\u003e\nDate:   Fri Sep 23 16:55:49 2016 -0400\n\n    cgroup: fix invalid controller enable rejections with cgroup namespace\n\n    On the v2 hierarchy, \"cgroup.subtree_control\" rejects controller\n    enables if the cgroup has processes in it.  The enforcement of this\n    logic assumes that the cgroup wouldn\u0027t have any css_sets associated\n    with it if there are no tasks in the cgroup, which is no longer true\n    since a79a908fd2b0 (\"cgroup: introduce cgroup namespaces\").\n\n    When a cgroup namespace is created, it pins the css_set of the\n    creating task to use it as the root css_set of the namespace.  This\n    extra reference stays as long as the namespace is around and makes\n    \"cgroup.subtree_control\" think that the namespace root cgroup is not\n    empty even when it is and thus reject controller enables.\n\n    Fix it by making cgroup_subtree_control() walk and test emptiness of\n    each css_set instead of testing whether the list_head is empty.\n\n    While at it, update the comment of cgroup_task_count() to indicate\n    that the returned value may be higher than the number of tasks, which\n    has always been true due to temporary references and doesn\u0027t break\n    anything.\n\n    Signed-off-by: Tejun Heo \u003ctj@kernel.org\u003e\n    Reported-by: Evgeny Vereshchagin \u003cevvers@ya.ru\u003e\n    Cc: Serge E. Hallyn \u003cserge.hallyn@ubuntu.com\u003e\n    Cc: Aditya Kali \u003cadityakali@google.com\u003e\n    Cc: Eric W. Biederman \u003cebiederm@xmission.com\u003e\n    Cc: stable@vger.kernel.org # v4.6+\n    Fixes: a79a908fd2b0 (\"cgroup: introduce cgroup namespaces\")\n    Link: https://github.com/systemd/systemd/pull/3589#issuecomment-249089541\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 3fb86dba27fc70396927aac17601dbd717a60eb3\nAuthor: Andrey Vagin \u003cavagin@openvz.org\u003e\nDate:   Tue Sep 6 00:47:13 2016 -0700\n\n    kernel: add a helper to get an owning user namespace for a namespace\n\n    Return -EPERM if an owning user namespace is outside of a process\n    current user namespace.\n\n    v2: In a first version ns_get_owner returned ENOENT for init_user_ns.\n        This special cases was removed from this version. There is nothing\n        outside of init_user_ns, so we can return EPERM.\n    v3: rename ns-\u003eget_owner() to ns-\u003eowner(). get_* usually means that it\n    grabs a reference.\n\n    Acked-by: Serge Hallyn \u003cserge@hallyn.com\u003e\n    Signed-off-by: Andrei Vagin \u003cavagin@openvz.org\u003e\n    Signed-off-by: Eric W. Biederman \u003cebiederm@xmission.com\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit e544e09109574f39595e7b069df839c857c6302d\nAuthor: Seth Forshee \u003cseth.forshee@canonical.com\u003e\nDate:   Wed Sep 23 15:16:04 2015 -0500\n\n    fs: Limit file caps to the user namespace of the super block\n\n    Capability sets attached to files must be ignored except in the\n    user namespaces where the mounter is privileged, i.e. s_user_ns\n    and its descendants. Otherwise a vector exists for gaining\n    privileges in namespaces where a user is not already privileged.\n\n    Add a new helper function, current_in_user_ns(), to test whether a user\n    namespace is the same as or a descendant of another namespace.\n    Use this helper to determine whether a file\u0027s capability set\n    should be applied to the caps constructed during exec.\n\n    --EWB Replaced in_userns with the simpler current_in_userns.\n\n    Acked-by: Serge Hallyn \u003cserge.hallyn@canonical.com\u003e\n    Signed-off-by: Seth Forshee \u003cseth.forshee@canonical.com\u003e\n    Signed-off-by: Eric W. Biederman \u003cebiederm@xmission.com\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 4ec041662bc02b09e745c6b69b7891ee1478c9b1\nAuthor: Johannes Weiner \u003cjweiner@fb.com\u003e\nDate:   Mon Sep 19 14:44:38 2016 -0700\n\n    cgroup: duplicate cgroup reference when cloning sockets\n\n    When a socket is cloned, the associated sock_cgroup_data is duplicated\n    but not its reference on the cgroup.  As a result, the cgroup reference\n    count will underflow when both sockets are destroyed later on.\n\n    Fixes: bd1060a1d671 (\"sock, cgroup: add sock-\u003esk_cgroup\")\n    Link: http://lkml.kernel.org/r/20160914194846.11153-2-hannes@cmpxchg.org\n    Signed-off-by: Johannes Weiner \u003channes@cmpxchg.org\u003e\n    Acked-by: Tejun Heo \u003ctj@kernel.org\u003e\n    Cc: Michal Hocko \u003cmhocko@suse.cz\u003e\n    Cc: Vladimir Davydov \u003cvdavydov@virtuozzo.com\u003e\n    Cc: \u003cstable@vger.kernel.org\u003e\t[4.5+]\n    Signed-off-by: Andrew Morton \u003cakpm@linux-foundation.org\u003e\n    Signed-off-by: Linus Torvalds \u003ctorvalds@linux-foundation.org\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 73c0e5722e9fc6f1b7fa2058fd083ceaae2568b2\nAuthor: Tejun Heo \u003ctj@kernel.org\u003e\nDate:   Wed Aug 10 11:23:44 2016 -0400\n\n    cgroup: make cgroup_path() and friends behave in the style of strlcpy()\n\n    cgroup_path() and friends used to format the path from the end and\n    thus the resulting path usually didn\u0027t start at the start of the\n    passed in buffer.  Also, when the buffer was too small, the partial\n    result was truncated from the head rather than tail and there was no\n    way to tell how long the full path would be.  These make the functions\n    less robust and more awkward to use.\n\n    With recent updates to kernfs_path(), cgroup_path() and friends can be\n    made to behave in strlcpy() style.\n\n    * cgroup_path(), cgroup_path_ns[_locked]() and task_cgroup_path() now\n      always return the length of the full path.  If buffer is too small,\n      it contains nul terminated truncated output.\n\n    * All users updated accordingly.\n\n    v2: cgroup_path() usage in kernel/sched/debug.c converted.\n\n    Signed-off-by: Tejun Heo \u003ctj@kernel.org\u003e\n    Acked-by: Greg Kroah-Hartman \u003cgregkh@linuxfoundation.org\u003e\n    Cc: Serge Hallyn \u003cserge.hallyn@ubuntu.com\u003e\n    Cc: Peter Zijlstra \u003cpeterz@infradead.org\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n    Change-Id: I8f16f5cb47c1eae59ad7539ea3fa3af825e0a125\n\ncommit d305d57e20372eb84ec7fdad134eb7c77238bd7d\nAuthor: Eric W. Biederman \u003cebiederm@xmission.com\u003e\nDate:   Fri Jul 15 06:36:44 2016 -0500\n\n    cgroupns: Only allow creation of hierarchies in the initial cgroup namespace\n\n    Unprivileged users can\u0027t use hierarchies if they create them as they do not\n    have privilieges to the root directory.\n\n    Which means the only thing a hiearchy created by an unprivileged user\n    is good for is expanding the number of cgroup links in every css_set,\n    which is a DOS attack.\n\n    We could allow hierarchies to be created in namespaces in the initial\n    user namespace.  Unfortunately there is only a single namespace for\n    the names of heirarchies, so that is likely to create more confusion\n    than not.\n\n    So do the simple thing and restrict hiearchy creation to the initial\n    cgroup namespace.\n\n    Cc: stable@vger.kernel.org\n    Fixes: a79a908fd2b0 (\"cgroup: introduce cgroup namespaces\")\n    Signed-off-by: \"Eric W. Biederman\" \u003cebiederm@xmission.com\u003e\n    Signed-off-by: Tejun Heo \u003ctj@kernel.org\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 5fbfc280f10ea9af918a7429e831b42d81d58860\nAuthor: Eric W. Biederman \u003cebiederm@xmission.com\u003e\nDate:   Fri Jul 15 06:35:24 2016 -0500\n\n    cgroupns: Fix the locking in copy_cgroup_ns\n\n    If \"clone(CLONE_NEWCGROUP...)\" is called it results in a nice lockdep\n    valid splat.\n\n    In __cgroup_proc_write the lock ordering is:\n         cgroup_mutex -- through cgroup_kn_lock_live\n         cgroup_threadgroup_rwsem\n\n    In copy_process the guts of clone the lock ordering is:\n         cgroup_threadgroup_rwsem -- through threadgroup_change_begin\n         cgroup_mutex -- through copy_namespaces -- copy_cgroup_ns\n\n    lockdep reports some a different call chains for the first ordering of\n    cgroup_mutex and cgroup_threadgroup_rwsem but it is harder to trace.\n    This is most definitely deadlock potential under the right\n    circumstances.\n\n    Fix this by by skipping the cgroup_mutex and making the locking in\n    copy_cgroup_ns mirror the locking in cgroup_post_fork which also runs\n    during fork under the cgroup_threadgroup_rwsem.\n\n    Cc: stable@vger.kernel.org\n    Fixes: a79a908fd2b0 (\"cgroup: introduce cgroup namespaces\")\n    Signed-off-by: \"Eric W. Biederman\" \u003cebiederm@xmission.com\u003e\n    Signed-off-by: Tejun Heo \u003ctj@kernel.org\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 3fa46142f591815dbaf2c860b356661755991aab\nAuthor: Martin KaFai Lau \u003ckafai@fb.com\u003e\nDate:   Thu Jun 30 10:28:42 2016 -0700\n\n    cgroup: Add cgroup_get_from_fd\n\n    Add a helper function to get a cgroup2 from a fd.  It will be\n    stored in a bpf array (BPF_MAP_TYPE_CGROUP_ARRAY) which will\n    be introduced in the later patch.\n\n    Signed-off-by: Martin KaFai Lau \u003ckafai@fb.com\u003e\n    Cc: Alexei Starovoitov \u003cast@fb.com\u003e\n    Cc: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Cc: Tejun Heo \u003ctj@kernel.org\u003e\n    Acked-by: Tejun Heo \u003ctj@kernel.org\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 8eb20de623d2df379b3a12c3ef34f86706c7cf9c\nAuthor: Tejun Heo \u003ctj@kernel.org\u003e\nDate:   Tue Jun 21 13:06:24 2016 -0400\n\n    cgroup: allow NULL return from ss-\u003ecss_alloc()\n\n    cgroup core expected css_alloc to return an ERR_PTR value on failure\n    and caused NULL deref if it returned NULL.  It\u0027s an easy mistake to\n    make from an alloc function and there\u0027s no ambiguity in what\u0027s being\n    indicated.  Update css_create() so that it interprets NULL return from\n    css_alloc as -ENOMEM.\n\n    Signed-off-by: Tejun Heo \u003ctj@kernel.org\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 61856241adaedfc4a0ecb161a492b2c34c52d9e8\nAuthor: Johannes Weiner \u003channes@cmpxchg.org\u003e\nDate:   Fri Jun 17 12:24:27 2016 -0400\n\n    cgroup: remove unnecessary 0 check from css_from_id()\n\n    css_idr allocation starts at 1, so index 0 will never point to an\n    item. css_from_id() currently filters that before asking idr_find(),\n    but idr_find() would also just return NULL, so this is not needed.\n\n    Signed-off-by: Johannes Weiner \u003channes@cmpxchg.org\u003e\n    Signed-off-by: Tejun Heo \u003ctj@kernel.org\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit e25a3db33d61ef2d8c828928eb5176bcc726bc46\nAuthor: Johannes Weiner \u003channes@cmpxchg.org\u003e\nDate:   Fri Jun 17 12:23:59 2016 -0400\n\n    cgroup: fix idr leak for the first cgroup root\n\n    The valid cgroup hierarchy ID range includes 0, so we can\u0027t filter for\n    positive numbers when freeing it, or it\u0027ll leak the first ID. No big\n    deal, just disruptive when reading the code.\n\n    The ID is freed during error handling and when the reference count\n    hits zero, so the double-free test is not necessary; remove it.\n\n    Signed-off-by: Johannes Weiner \u003channes@cmpxchg.org\u003e\n    Signed-off-by: Tejun Heo \u003ctj@kernel.org\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 45875d6655e72fb2517b44547bc1a72847f89d1c\nAuthor: Wenwei Tao \u003cww.tao0320@gmail.com\u003e\nDate:   Fri May 13 22:59:20 2016 +0800\n\n    cgroup: remove redundant cleanup in css_create\n\n    When create css failed, before call css_free_rcu_fn, we remove the css\n    id and exit the percpu_ref, but we will do these again in\n    css_free_work_fn, so they are redundant.  Especially the css id, that\n    would cause problem if we remove it twice, since it may be assigned to\n    another css after the first remove.\n\n    tj: This was broken by two commits updating the free path without\n        synchronizing the creation failure path.  This can be easily\n        triggered by trying to create more than 64k memory cgroups.\n\n    Signed-off-by: Wenwei Tao \u003cww.tao0320@gmail.com\u003e\n    Signed-off-by: Tejun Heo \u003ctj@kernel.org\u003e\n    Cc: Vladimir Davydov \u003cvdavydov@parallels.com\u003e\n    Fixes: 9a1049da9bd2 (\"percpu-refcount: require percpu_ref to be exited explicitly\")\n    Fixes: 01e586598b22 (\"cgroup: release css-\u003eid after css_free\")\n    Cc: stable@vger.kernel.org # v3.17+\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 06f9f5f260b77cb9d9536d1bcdcfb4251a43fb6b\nAuthor: Felipe Balbi \u003cfelipe.balbi@linux.intel.com\u003e\nDate:   Thu May 12 12:34:38 2016 +0300\n\n    cgroup: fix compile warning\n\n    commit 4f41fc59620f (\"cgroup, kernfs: make mountinfo\n     show properly scoped path for cgroup namespaces\")\n     added the following compile warning:\n\n    kernel/cgroup.c: In function ‘cgroup_show_path’:\n    kernel/cgroup.c:1634:15: warning: unused variable ‘ret’ [-Wunused-variable]\n      int len \u003d 0, ret \u003d 0;\n                   ^\n    fix it.\n\n    Fixes: 4f41fc59620f (\"cgroup, kernfs: make mountinfo show properly scoped path for cgroup namespaces\")\n    Signed-off-by: Felipe Balbi \u003cfelipe.balbi@linux.intel.com\u003e\n    Signed-off-by: Tejun Heo \u003ctj@kernel.org\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 3887afe1843a8e67b65ba79803ece0488b76c3d9\nAuthor: Serge E. Hallyn \u003cserge.hallyn@ubuntu.com\u003e\nDate:   Mon May 9 09:59:55 2016 -0500\n\n    cgroup, kernfs: make mountinfo show properly scoped path for cgroup namespaces\n\n    Patch summary:\n\n    When showing a cgroupfs entry in mountinfo, show the path of the mount\n    root dentry relative to the reader\u0027s cgroup namespace root.\n\n    Short explanation (courtesy of mkerrisk):\n\n    If we create a new cgroup namespace, then we want both /proc/self/cgroup\n    and /proc/self/mountinfo to show cgroup paths that are correctly\n    virtualized with respect to the cgroup mount point.  Previous to this\n    patch, /proc/self/cgroup shows the right info, but /proc/self/mountinfo\n    does not.\n\n    Long version:\n\n    When a uid 0 task which is in freezer cgroup /a/b, unshares a new cgroup\n    namespace, and then mounts a new instance of the freezer cgroup, the new\n    mount will be rooted at /a/b.  The root dentry field of the mountinfo\n    entry will show \u0027/a/b\u0027.\n\n     cat \u003e /tmp/do1 \u003c\u003c EOF\n     mount -t cgroup -o freezer freezer /mnt\n     grep freezer /proc/self/mountinfo\n     EOF\n\n     unshare -Gm  bash /tmp/do1\n     \u003e 330 160 0:34 / /sys/fs/cgroup/freezer rw,nosuid,nodev,noexec,relatime - cgroup cgroup rw,freezer\n     \u003e 355 133 0:34 /a/b /mnt rw,relatime - cgroup freezer rw,freezer\n\n    The task\u0027s freezer cgroup entry in /proc/self/cgroup will simply show\n    \u0027/\u0027:\n\n     grep freezer /proc/self/cgroup\n     9:freezer:/\n\n    If instead the same task simply bind mounts the /a/b cgroup directory,\n    the resulting mountinfo entry will again show /a/b for the dentry root.\n    However in this case the task will find its own cgroup at /mnt/a/b,\n    not at /mnt:\n\n     mount --bind /sys/fs/cgroup/freezer/a/b /mnt\n     130 25 0:34 /a/b /mnt rw,nosuid,nodev,noexec,relatime shared:21 - cgroup cgroup rw,freezer\n\n    In other words, there is no way for the task to know, based on what is\n    in mountinfo, which cgroup directory is its own.\n\n    Example (by mkerrisk):\n\n    First, a little script to save some typing and verbiage:\n\n    echo -e \"\\t/proc/self/cgroup:\\t$(cat /proc/self/cgroup | grep freezer)\"\n    cat /proc/self/mountinfo | grep freezer |\n            awk \u0027{print \"\\tmountinfo:\\t\\t\" $4 \"\\t\" $5}\u0027\n\n    Create cgroup, place this shell into the cgroup, and look at the state\n    of the /proc files:\n\n    2653\n    2653                         # Our shell\n    14254                        # cat(1)\n            /proc/self/cgroup:      10:freezer:/a/b\n            mountinfo:              /       /sys/fs/cgroup/freezer\n\n    Create a shell in new cgroup and mount namespaces. The act of creating\n    a new cgroup namespace causes the process\u0027s current cgroups directories\n    to become its cgroup root directories. (Here, I\u0027m using my own version\n    of the \"unshare\" utility, which takes the same options as the util-linux\n    version):\n\n    Look at the state of the /proc files:\n\n            /proc/self/cgroup:      10:freezer:/\n            mountinfo:              /       /sys/fs/cgroup/freezer\n\n    The third entry in /proc/self/cgroup (the pathname of the cgroup inside\n    the hierarchy) is correctly virtualized w.r.t. the cgroup namespace, which\n    is rooted at /a/b in the outer namespace.\n\n    However, the info in /proc/self/mountinfo is not for this cgroup\n    namespace, since we are seeing a duplicate of the mount from the\n    old mount namespace, and the info there does not correspond to the\n    new cgroup namespace. However, trying to create a new mount still\n    doesn\u0027t show us the right information in mountinfo:\n\n                                          # propagating to other mountns\n            /proc/self/cgroup:      7:freezer:/\n            mountinfo:              /a/b    /mnt/freezer\n\n    The act of creating a new cgroup namespace caused the process\u0027s\n    current freezer directory, \"/a/b\", to become its cgroup freezer root\n    directory. In other words, the pathname directory of the directory\n    within the newly mounted cgroup filesystem should be \"/\",\n    but mountinfo wrongly shows us \"/a/b\". The consequence of this is\n    that the process in the cgroup namespace cannot correctly construct\n    the pathname of its cgroup root directory from the information in\n    /proc/PID/mountinfo.\n\n    With this patch, the dentry root field in mountinfo is shown relative\n    to the reader\u0027s cgroup namespace.  So the same steps as above:\n\n            /proc/self/cgroup:      10:freezer:/a/b\n            mountinfo:              /       /sys/fs/cgroup/freezer\n            /proc/self/cgroup:      10:freezer:/\n            mountinfo:              /../..  /sys/fs/cgroup/freezer\n            /proc/self/cgroup:      10:freezer:/\n            mountinfo:              /       /mnt/freezer\n\n    cgroup.clone_children  freezer.parent_freezing  freezer.state      tasks\n    cgroup.procs           freezer.self_freezing    notify_on_release\n    3164\n    2653                   # First shell that placed in this cgroup\n    3164                   # Shell started by \u0027unshare\u0027\n    14197                  # cat(1)\n\n    Signed-off-by: Serge Hallyn \u003cserge.hallyn@ubuntu.com\u003e\n    Tested-by: Michael Kerrisk \u003cmtk.manpages@gmail.com\u003e\n    Acked-by: Michael Kerrisk \u003cmtk.manpages@gmail.com\u003e\n    Signed-off-by: Tejun Heo \u003ctj@kernel.org\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 7b19bf36d0103807b0f478e5523c6f6e539a781a\nAuthor: Tejun Heo \u003ctj@kernel.org\u003e\nDate:   Tue Mar 8 11:51:26 2016 -0500\n\n    cgroup: implement cgroup_subsys-\u003eimplicit_on_dfl\n\n    Some controllers, perf_event for now and possibly freezer in the\n    future, don\u0027t really make sense to control explicitly through\n    \"cgroup.subtree_control\".  For example, the primary role of perf_event\n    is identifying the cgroups of tasks; however, because the controller\n    also keeps a small amount of state per cgroup, it can\u0027t be replaced\n    with simple cgroup membership tests.\n\n    This patch implements cgroup_subsys-\u003eimplicit_on_dfl flag.  When set,\n    the controller is implicitly enabled on all cgroups on the v2\n    hierarchy so that utility type controllers such as perf_event can be\n    enabled and function transparently.\n\n    An implicit controller doesn\u0027t show up in \"cgroup.controllers\" or\n    \"cgroup.subtree_control\", is exempt from no internal process rule and\n    can be stolen from the default hierarchy even if there are non-root\n    csses.\n\n    v2: Reimplemented on top of the recent updates to css handling and\n        subsystem rebinding.  Rebinding implicit subsystems is now a\n        simple matter of exempting it from the busy subsystem check.\n\n    Signed-off-by: Tejun Heo \u003ctj@kernel.org\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit f67ca4f794d2a49c7ee498716012ac958dd843d0\nAuthor: Tejun Heo \u003ctj@kernel.org\u003e\nDate:   Tue Mar 8 11:51:26 2016 -0500\n\n    cgroup: use css_set-\u003emg_dst_cgrp for the migration target cgroup\n\n    Migration can be multi-target on the default hierarchy when a\n    controller is enabled - processes belonging to each child cgroup have\n    to be moved to the child cgroup itself to refresh css association.\n\n    This isn\u0027t a problem for cgroup_migrate_add_src() as each source\n    css_set still maps to single source and target cgroups; however,\n    cgroup_migrate_prepare_dst() is called once after all source css_sets\n    are added and thus might not have a single destination cgroup.  This\n    is currently worked around by specifying NULL for @dst_cgrp and using\n    the source\u0027s default cgroup as destination as the only multi-target\n    migration in use is self-targetting.  While this works, it\u0027s subtle\n    and clunky.\n\n    As all taget cgroups are already specified while preparing the source\n    css_sets, this clunkiness can easily be removed by recording the\n    target cgroup in each source css_set.  This patch adds\n    css_set-\u003emg_dst_cgrp which is recorded on cgroup_migrate_src() and\n    used by cgroup_migrate_prepare_dst().  This also makes migration code\n    ready for arbitrary multi-target migration.\n\n    Signed-off-by: Tejun Heo \u003ctj@kernel.org\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 5a298565a2b491136dba1f06e753eb37d1cf5c77\nAuthor: Tejun Heo \u003ctj@kernel.org\u003e\nDate:   Tue Mar 8 11:51:26 2016 -0500\n\n    cgroup: make cgroup[_taskset]_migrate() take cgroup_root instead of cgroup\n\n    On the default hierarchy, a migration can be multi-source and/or\n    multi-destination.  cgroup_taskest_migrate() used to incorrectly\n    assume single destination cgroup but the bug has been fixed by\n    1f7dd3e5a6e4 (\"cgroup: fix handling of multi-destination migration\n    from subtree_control enabling\").\n\n    Since the commit, @dst_cgrp to cgroup[_taskset]_migrate() is only used\n    to determine which subsystems are affected or which cgroup_root the\n    migration is taking place in.  As such, @dst_cgrp is misleading.  This\n    patch replaces @dst_cgrp with @root.\n\n    Signed-off-by: Tejun Heo \u003ctj@kernel.org\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 5c9ddbc98a00396e6f984563db930519f8af27a7\nAuthor: Tejun Heo \u003ctj@kernel.org\u003e\nDate:   Tue Mar 8 11:51:25 2016 -0500\n\n    cgroup: move migration destination verification out of cgroup_migrate_prepare_dst()\n\n    cgroup_migrate_prepare_dst() verifies whether the destination cgroup\n    is allowable; however, the test doesn\u0027t really belong there.  It\u0027s too\n    deep and common in the stack and as a result the test itself is gated\n    by another test.\n\n    Separate the test out into cgroup_may_migrate_to() and update\n    cgroup_attach_task() and cgroup_transfer_tasks() to perform the test\n    directly.  This doesn\u0027t cause any behavior differences.\n\n    Signed-off-by: Tejun Heo \u003ctj@kernel.org\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 5a3d735de7547f062bf71c10299fe3461b3f3bf1\nAuthor: Tejun Heo \u003ctj@kernel.org\u003e\nDate:   Tue Mar 8 11:51:25 2016 -0500\n\n    cgroup: fix incorrect destination cgroup in cgroup_update_dfl_csses()\n\n    cgroup_update_dfl_csses() should move each task in the subtree to\n    self; however, it was incorrectly calling cgroup_migrate_add_src()\n    with the root of the subtree as @dst_cgrp.  Fortunately,\n    cgroup_migrate_add_src() currently uses @dst_cgrp only to determine\n    the hierarchy and the bug doesn\u0027t cause any actual breakages.  Fix it.\n\n    Signed-off-by: Tejun Heo \u003ctj@kernel.org\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit f9017131600b76950759247ff5d93f7b50873d2c\nAuthor: Tejun Heo \u003ctj@kernel.org\u003e\nDate:   Thu Mar 3 09:58:01 2016 -0500\n\n    cgroup: update css iteration in cgroup_update_dfl_csses()\n\n    The existing sequences of operations ensure that the offlining csses\n    are drained before cgroup_update_dfl_csses(), so even though\n    cgroup_update_dfl_csses() uses css_for_each_descendant_pre() to walk\n    the target cgroups, it doesn\u0027t end up operating on dead cgroups.\n    Also, the function explicitly excludes the subtree root from\n    operation.\n\n    This is fragile and inconsistent with the rest of css update\n    operations.  This patch updates cgroup_update_dfl_csses() to use\n    cgroup_for_each_live_descendant_pre() instead and include the subtree\n    root.\n\n    Signed-off-by: Tejun Heo \u003ctj@kernel.org\u003e\n    Acked-by: Zefan Li \u003clizefan@huawei.com\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 3f82ae7e0edcddf742a23975cbb6ff9d97c11014\nAuthor: Tejun Heo \u003ctj@kernel.org\u003e\nDate:   Thu Mar 3 09:58:01 2016 -0500\n\n    cgroup: allocate 2x cgrp_cset_links when setting up a new root\n\n    During prep, cgroup_setup_root() allocates cgrp_cset_links matching\n    the number of existing css_sets to later link the new root.  This is\n    fine for now as the only operation which can happen inbetween is\n    rebind_subsystems() and rebinding of empty subsystems doesn\u0027t create\n    new css_sets.\n\n    However, while not yet allowed, with the recent reimplementation,\n    rebind_subsystems() can rebind subsystems with descendant csses and\n    thus can create new css_sets.  This patch makes cgroup_setup_root()\n    allocate 2x of the existing css_sets so that later use of live\n    subsystem rebinding doesn\u0027t blow up.\n\n    Signed-off-by: Tejun Heo \u003ctj@kernel.org\u003e\n    Acked-by: Zefan Li \u003clizefan@huawei.com\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 6faa69554235d709a0ccc86963f5abce29848701\nAuthor: Tejun Heo \u003ctj@kernel.org\u003e\nDate:   Thu Mar 3 09:58:01 2016 -0500\n\n    cgroup: make cgroup_calc_subtree_ss_mask() take @this_ss_mask\n\n    cgroup_calc_subtree_ss_mask() currently takes @cgrp and\n    @subtree_control.  @cgrp is used for two purposes - to decide whether\n    it\u0027s for default hierarchy and the mask of available subsystems.  The\n    former doesn\u0027t matter as the results are the same regardless.  The\n    latter can be specified directly through a subsystem mask.\n\n    This patch makes cgroup_calc_subtree_ss_mask() perform the same\n    calculations for both default and legacy hierarchies and take\n    @this_ss_mask for available subsystems.  @cgrp is no longer used and\n    dropped.  This is to allow using the function in contexts where\n    available controllers can\u0027t be decided from the cgroup.\n\n    v2: cgroup_refres_subtree_ss_mask() is removed by a previous patch.\n        Updated accordingly.\n\n    Signed-off-by: Tejun Heo \u003ctj@kernel.org\u003e\n    Acked-by: Zefan Li \u003clizefan@huawei.com\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit a6fa3a1ebe3b4873bb8afaf9078ab998eaf357ea\nAuthor: Tejun Heo \u003ctj@kernel.org\u003e\nDate:   Thu Mar 3 09:58:01 2016 -0500\n\n    cgroup: reimplement rebind_subsystems() using cgroup_apply_control() and friends\n\n    rebind_subsystem() open codes quite a bit of css and interface file\n    manipulations.  It tries to be fail-safe but doesn\u0027t quite achieve it.\n    It can be greatly simplified by using the new css management helpers.\n    This patch reimplements rebind_subsytsems() using\n    cgroup_apply_control() and friends.\n\n    * The half-baked rollback on file creation failure is dropped.  It is\n      an extremely cold path, failure isn\u0027t critical, and, aside from\n      kernel bugs, the only reason it can fail is memory allocation\n      failure which pretty much doesn\u0027t happen for small allocations.\n\n    * As cgroup_apply_control_disable() is now used to clean up root\n      cgroup on rebind, make sure that it doesn\u0027t end up killing root\n      csses.\n\n    * All callers of rebind_subsystems() are updated to use\n      cgroup_lock_and_drain_offline() as the apply_control functions\n      require drained subtree.\n\n    * This leaves cgroup_refresh_subtree_ss_mask() without any user.\n      Removed.\n\n    * css_populate_dir() and css_clear_dir() no longer needs\n      @cgrp_override parameter.  Dropped.\n\n    * While at it, add WARN_ON() to rebind_subsystem() calls which are\n      expected to always succeed just in case.\n\n    While the rules visible to userland aren\u0027t changed, this\n    reimplementation not only simplifies rebind_subsystems() but also\n    allows it to disable and enable csses recursively.  This can be used\n    to implement more flexible rebinding.\n\n    Signed-off-by: Tejun Heo \u003ctj@kernel.org\u003e\n    Acked-by: Zefan Li \u003clizefan@huawei.com\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit f640e4152984f8035a5f4d6b6cfe9ff42f7689ce\nAuthor: Tejun Heo \u003ctj@kernel.org\u003e\nDate:   Thu Mar 3 09:58:00 2016 -0500\n\n    cgroup: use cgroup_apply_enable_control() in cgroup creation path\n\n    cgroup_create() manually updates control masks and creates child csses\n    which cgroup_mkdir() then manually populates.  Both can be simplified\n    by using cgroup_apply_enable_control() and friends.  The only catch is\n    that it calls css_populate_dir() with NULL cgroup-\u003ekn during\n    cgroup_create().  This is worked around by making the function noop on\n    NULL kn.\n\n    Signed-off-by: Tejun Heo \u003ctj@kernel.org\u003e\n    Acked-by: Zefan Li \u003clizefan@huawei.com\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit c54842ae72c0a2ac5817af6a8b35b414aa277fe1\nAuthor: Tejun Heo \u003ctj@kernel.org\u003e\nDate:   Thu Mar 3 09:58:00 2016 -0500\n\n    cgroup: combine cgroup_mutex locking and offline css draining\n\n    cgroup_drain_offline() is used to wait for csses being offlined to\n    uninstall itself from cgroup-\u003esubsys[] array so that new csses can be\n    installed.  The function\u0027s only user, cgroup_subtree_control_write(),\n    calls it after performing some checks and restarts the whole process\n    via restart_syscall() if draining has to release cgroup_mutex to wait.\n\n    This can be simplified by draining before other synchronized\n    operations so that there\u0027s nothing to restart.  This patch converts\n    cgroup_drain_offline() to cgroup_lock_and_drain_offline() which\n    performs both locking and draining and updates cgroup_kn_lock_live()\n    use it instead of cgroup_mutex() if requested.  This combined locking\n    and draining operations are easier to use and less error-prone.\n\n    While at it, add WARNs in control_apply functions which triggers if\n    the subtree isn\u0027t properly drained.\n\n    Signed-off-by: Tejun Heo \u003ctj@kernel.org\u003e\n    Acked-by: Zefan Li \u003clizefan@huawei.com\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 625cd6484bc5cff232fab1cb45d29262ee9f5729\nAuthor: Tejun Heo \u003ctj@kernel.org\u003e\nDate:   Thu Mar 3 09:58:00 2016 -0500\n\n    cgroup: factor out cgroup_{apply|finalize}_control() from cgroup_subtree_control_write()\n\n    Factor out cgroup_{apply|finalize}_control() so that control mask\n    update can be done in several simple steps.  This patch doesn\u0027t\n    introduce behavior changes.\n\n    Signed-off-by: Tejun Heo \u003ctj@kernel.org\u003e\n    Acked-by: Zefan Li \u003clizefan@huawei.com\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 5bd05286da6f63d68ddcfb56cdf995ff1adf4ed3\nAuthor: Tejun Heo \u003ctj@kernel.org\u003e\nDate:   Thu Mar 3 09:57:59 2016 -0500\n\n    cgroup: introduce cgroup_{save|propagate|restore}_control()\n\n    While controllers are being enabled and disabled in\n    cgroup_subtree_control_write(), the original subsystem masks are\n    stashed in local variables so that they can be restored if the\n    operation fails in the middle.\n\n    This patch adds dedicated fields to struct cgroup to be used instead\n    of the local variables and implements functions to stash the current\n    values, propagate the changes and restore them recursively.  Combined\n    with the previous changes, this makes subsystem management operations\n    fully recursive and modularlized.  This will be used to expand cgroup\n    core functionalities.\n\n    While at it, remove now unused @css_enable and @css_disable from\n    cgroup_subtree_control_write().\n\n    Signed-off-by: Tejun Heo \u003ctj@kernel.org\u003e\n    Acked-by: Zefan Li \u003clizefan@huawei.com\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 4f1ca466ce38a618d6faa0119ea3cbbf2e45b582\nAuthor: Tejun Heo \u003ctj@kernel.org\u003e\nDate:   Thu Mar 3 09:57:59 2016 -0500\n\n    cgroup: make cgroup_drain_offline() and cgroup_apply_control_{disable|enable}() recursive\n\n    The three factored out css management operations -\n    cgroup_drain_offline() and cgroup_apply_control_{disable|enable}() -\n    only depend on the current state of the target cgroups and idempotent\n    and thus can be easily made to operate on the subtree instead of the\n    immediate children.\n\n    This patch introduces the iterators which walk live subtree and\n    converts the three functions to operate on the subtree including self\n    instead of the children.  While this leads to spurious walking and be\n    slightly more expensive, it will allow them to be used for wider scope\n    of operations.\n\n    Note that cgroup_drain_offline() now tests for whether a css is dying\n    before trying to drain it.  This is to avoid trying to drain live\n    csses as there can be mix of live and dying csses in a subtree unlike\n    children of the same parent.\n\n    Signed-off-by: Tejun Heo \u003ctj@kernel.org\u003e\n    Acked-by: Zefan Li \u003clizefan@huawei.com\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 1618a596e9d8624aa4dca92a9b1b8d3ea02523f4\nAuthor: Tejun Heo \u003ctj@kernel.org\u003e\nDate:   Thu Mar 3 09:57:59 2016 -0500\n\n    cgroup: factor out cgroup_apply_control_enable() from cgroup_subtree_control_write()\n\n    Factor out css enabling and showing into cgroup_apply_control_enable().\n\n    * Nest subsystem walk inside child walk.  The child walk will later be\n      converted to subtree walk which is a bit more expensive.\n\n    * Instead of operating on the differential masks @css_enable, simply\n      enable or show csses according to the current cgroup_control() and\n      cgroup_ss_mask().  This leads to the same result and is simpler and\n      more robust.\n\n    Signed-off-by: Tejun Heo \u003ctj@kernel.org\u003e\n    Acked-by: Zefan Li \u003clizefan@huawei.com\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit a1a9db1ad261c6cd33aad5980aaad3a985957bd5\nAuthor: Tejun Heo \u003ctj@kernel.org\u003e\nDate:   Thu Mar 3 09:57:59 2016 -0500\n\n    cgroup: factor out cgroup_apply_control_disable() from cgroup_subtree_control_write()\n\n    Factor out css disabling and hiding into cgroup_apply_control_disable().\n\n    * Nest subsystem walk inside child walk.  The child walk will later be\n      converted to subtree walk which is a bit more expensive.\n\n    * Instead of operating on the differential masks @css_enable and\n      @css_disable, simply disable or hide csses according to the current\n      cgroup_control() and cgroup_ss_mask().  This leads to the same\n      result and is simpler and more robust.\n\n    * This allows error handling path to share the same code.\n\n    Signed-off-by: Tejun Heo \u003ctj@kernel.org\u003e\n    Acked-by: Zefan Li \u003clizefan@huawei.com\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 59c06cb28d873b12b6e161366783b318db208912\nAuthor: Tejun Heo \u003ctj@kernel.org\u003e\nDate:   Thu Mar 3 09:57:59 2016 -0500\n\n    cgroup: factor out cgroup_drain_offline() from cgroup_subtree_control_write()\n\n    Factor out async css offline draining into cgroup_drain_offline().\n\n    * Nest subsystem walk inside child walk.  The child walk will later be\n      converted to subtree walk which is a bit more expensive.\n\n    * Relocate the draining above subsystem mask preparation, which\n      doesn\u0027t create any behavior differences but helps further\n      refactoring.\n\n    Signed-off-by: Tejun Heo \u003ctj@kernel.org\u003e\n    Acked-by: Zefan Li \u003clizefan@huawei.com\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 4083639cd15ab9941a5cb65ef04ce325203e124f\nAuthor: Tejun Heo \u003ctj@kernel.org\u003e\nDate:   Thu Mar 3 09:57:58 2016 -0500\n\n    cgroup: introduce cgroup_control() and cgroup_ss_mask()\n\n    When a controller is enabled and visible on a non-root cgroup is\n    determined by subtree_control and subtree_ss_mask of the parent\n    cgroup.  For a root cgroup, by the type of the hierarchy and which\n    controllers are attached to it.  Deciding the above on each usage is\n    fragile and unnecessarily complicates the users.\n\n    This patch introduces cgroup_control() and cgroup_ss_mask() which\n    calculate and return the [visibly] enabled subsyste mask for the\n    specified cgroup and conver the existing usages.\n\n    * cgroup_e_css() is restructured for simplicity.\n\n    * cgroup_calc_subtree_ss_mask() and cgroup_subtree_control_write() no\n      longer need to distinguish root and non-root cases.\n\n    * With cgroup_control(), cgroup_controllers_show() can now handle both\n      root and non-root cases.  cgroup_root_controllers_show() is removed.\n\n    v2: cgroup_control() updated to yield the correct result on v1\n        hierarchies too.  cgroup_subtree_control_write() converted.\n\n    Signed-off-by: Tejun Heo \u003ctj@kernel.org\u003e\n    Acked-by: Zefan Li \u003clizefan@huawei.com\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 7fd605c7bc679e356623744469b00f094ecf7729\nAuthor: Tejun Heo \u003ctj@kernel.org\u003e\nDate:   Thu Mar 3 09:57:58 2016 -0500\n\n    cgroup: factor out cgroup_create() out of cgroup_mkdir()\n\n    We\u0027re in the process of refactoring cgroup and css management paths to\n    separate them out to eventually allow cgroups which aren\u0027t visible\n    through cgroup fs.  This patch factors out cgroup_create() out of\n    cgroup_mkdir().  cgroup_create() contains all internal object creation\n    and initialization.  cgroup_mkdir() uses cgroup_create() to create the\n    internal cgroup and adds interface directory and file creation.\n\n    This patch doesn\u0027t cause any behavior differences.\n\n    Signed-off-by: Tejun Heo \u003ctj@kernel.org\u003e\n    Acked-by: Zefan Li \u003clizefan@huawei.com\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 7123b2ea7428e057a3192c18a20ba6455699491c\nAuthor: Tejun Heo \u003ctj@kernel.org\u003e\nDate:   Thu Mar 3 09:57:58 2016 -0500\n\n    cgroup: reorder operations in cgroup_mkdir()\n\n    Currently, operations to initialize internal objects and create\n    interface directory and files are intermixed in cgroup_mkdir().  We\u0027re\n    in the process of refactoring cgroup and css management paths to\n    separate them out to eventually allow cgroups which aren\u0027t visible\n    through cgroup fs.\n\n    This patch reorders operations inside cgroup_mkdir() so that interface\n    directory and file handling comes after internal object\n    initialization.  This will enable further refactoring.\n\n    Signed-off-by: Tejun Heo \u003ctj@kernel.org\u003e\n    Acked-by: Zefan Li \u003clizefan@huawei.com\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 2066c2c2ede38412e445a4d66388513f583deca4\nAuthor: Tejun Heo \u003ctj@kernel.org\u003e\nDate:   Thu Mar 3 09:57:58 2016 -0500\n\n    cgroup: explicitly track whether a cgroup_subsys_state is visible to userland\n\n    Currently, whether a css (cgroup_subsys_state) has its interface files\n    created is not tracked and assumed to change together with the owning\n    cgroup\u0027s lifecycle.  cgroup directory and interface creation is being\n    separated out from internal object creation to help refactoring and\n    eventually allow cgroups which are not visible through cgroupfs.\n\n    This patch adds CSS_VISIBLE to track whether a css has its interface\n    files created and perform management operations only when necessary\n    which helps decoupling interface file handling from internal object\n    lifecycle.  After this patch, all css interface file management\n    functions can be called regardless of the current state and will\n    achieve the expected result.\n\n    Signed-off-by: Tejun Heo \u003ctj@kernel.org\u003e\n    Acked-by: Zefan Li \u003clizefan@huawei.com\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit b0318e7d530f2047434e44b4b48f1c79bb3f1bb6\nAuthor: Tejun Heo \u003ctj@kernel.org\u003e\nDate:   Thu Mar 3 09:57:58 2016 -0500\n\n    cgroup: separate out interface file creation from css creation\n\n    Currently, interface files are created when a css is created depending\n    on whether @visible is set.  This patch separates out the two into\n    separate steps to help code refactoring and eventually allow cgroups\n    which aren\u0027t visible through cgroup fs.\n\n    Move css_populate_dir() out of create_css() and drop @visible.  While\n    at it, rename the function to css_create() for consistency.\n\n    Signed-off-by: Tejun Heo \u003ctj@kernel.org\u003e\n    Acked-by: Zefan Li \u003clizefan@huawei.com\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit f801c6eebaa5354f4d2e0af0c841410f409a7641\nAuthor: Tejun Heo \u003ctj@kernel.org\u003e\nDate:   Thu Mar 3 09:57:57 2016 -0500\n\n    cgroup: suppress spurious de-populated events\n\n    During task migration, tasks may transfer between two css_sets which\n    are associated with the same cgroup.  If those tasks are the only\n    tasks in the cgroup, this currently triggers a spurious de-populated\n    event on the cgroup.\n\n    Fix it by bumping up populated count before bumping it down during\n    migration to ensure that it doesn\u0027t reach zero spuriously.\n\n    Signed-off-by: Tejun Heo \u003ctj@kernel.org\u003e\n    Acked-by: Zefan Li \u003clizefan@huawei.com\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit dd24f6c8c3ee76bb1941880bf569b7051d8b03c1\nAuthor: Tejun Heo \u003ctj@kernel.org\u003e\nDate:   Thu Mar 3 09:57:57 2016 -0500\n\n    cgroup: re-hash init_css_set after subsystems are initialized\n\n    css_sets are hashed by their subsys[] contents and in cgroup_init()\n    init_css_set is hashed early, before subsystem inits, when all entries\n    in its subsys[] are NULL, so that cgroup_dfl_root initialization can\n    find and link to it.  As subsystems are initialized,\n    init_css_set.subsys[] is filled up but the hashing is never updated\n    making init_css_set hashed in the wrong place.  While incorrect, this\n    doesn\u0027t cause a critical failure as css_set management code would\n    create an identical css_set dynamically.\n\n    Fix it by rehashing init_css_set after subsystems are initialized.\n    While at it, drop unnecessary @key local variable.\n\n    Signed-off-by: Tejun Heo \u003ctj@kernel.org\u003e\n    Acked-by: Zefan Li \u003clizefan@huawei.com\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 2744bd880e5ff4aa7541856f1c6ab24e684bfd26\nAuthor: Vladimir Davydov \u003cvdavydov@virtuozzo.com\u003e\nDate:   Tue Mar 1 19:56:30 2016 +0300\n\n    cgroup: reset css on destruction\n\n    An associated css can be around for quite a while after a cgroup\n    directory has been removed. In general, it makes sense to reset it to\n    defaults so as not to worry about any remnants. For instance, memory\n    cgroup needs to reset memory.low, otherwise pages charged to a dead\n    cgroup might never get reclaimed. There\u0027s -\u003ecss_reset callback, which\n    would fit perfectly for the purpose. Currently, it\u0027s only called when a\n    subsystem is disabled in the unified hierarchy and there are other\n    subsystems dependant on it. Let\u0027s call it on css destruction as well.\n\n    Suggested-by: Johannes Weiner \u003channes@cmpxchg.org\u003e\n    Signed-off-by: Vladimir Davydov \u003cvdavydov@virtuozzo.com\u003e\n    Signed-off-by: Tejun Heo \u003ctj@kernel.org\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit f5bd5544b6bddb53f7b5d308484217fbc45aa33e\nAuthor: Tejun Heo \u003ctj@kernel.org\u003e\nDate:   Sun Feb 28 08:59:33 2016 -0500\n\n    cgroup: fix and restructure error handling in copy_cgroup_ns()\n\n    copy_cgroup_ns()\u0027s error handling was broken and the attempt to fix it\n    d22025570e2e (\"cgroup: fix alloc_cgroup_ns() error handling in\n    copy_cgroup_ns()\") was broken too in that it ended up trying an\n    ERR_PTR() value.\n\n    There\u0027s only one place where copy_cgroup_ns() needs to perform cleanup\n    after failure.  Simplify and fix the error handling by removing the\n    goto\u0027s.\n\n    Signed-off-by: Tejun Heo \u003ctj@kernel.org\u003e\n    Reported-by: Dan Carpenter \u003cdan.carpenter@oracle.com\u003e\n    Acked-by: Serge E. Hallyn \u003cserge.hallyn@ubuntu.com\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 4c9059210bbb9f70f40477b4646a068d64f6ba5a\nAuthor: Xiubo Li \u003clixiubo@cmss.chinamobile.com\u003e\nDate:   Fri Feb 26 13:07:38 2016 +0800\n\n    cgroup: fix a mistake in warning message\n\n    There is a mistake about the print format name:id \u003c--\u003e %d:%s, which\n    the name is \u0027char *\u0027 type and id is \u0027int\u0027 type.  Change \"name:id\" to\n    \"id:name\" instead to be consistent with \"cgroup_subsys %d:%s\".\n\n    Signed-off-by: Xiubo Li \u003clixiubo@cmss.chinamobile.com\u003e\n    Acked-by: Zefan Li \u003clizefan@huawei.com\u003e\n    Signed-off-by: Tejun Heo \u003ctj@kernel.org\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 4186b3fe864626b700ff6f88bb87d17be7d2f62e\nAuthor: Tejun Heo \u003ctj@kernel.org\u003e\nDate:   Tue Feb 23 10:00:51 2016 -0500\n\n    cgroup: use -\u003esubtree_control when testing no internal process rule\n\n    No internal process rule is enforced by cgroup_migrate_prepare_dst()\n    during process migration.  It tests whether the target cgroup\u0027s\n    -\u003echild_subsys_mask is zero which is different from \"subtree_control\"\n    write path which tests -\u003esubtree_control.  This hasn\u0027t mattered\n    because up until now, both -\u003echild_subsys_mask and -\u003esubtree_control\n    are zero or non-zero at the same time.  However, with the planned\n    addition of implicit controllers, this will no longer be true.\n\n    This patch prepares for the change by making\n    cgorup_migrate_prepare_dst() test -\u003esubtree_control instead.\n\n    Signed-off-by: Tejun Heo \u003ctj@kernel.org\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit b2318f76a5ebd42e4655cbdbfebab7c3aeef9355\nAuthor: Tejun Heo \u003ctj@kernel.org\u003e\nDate:   Tue Feb 23 10:00:51 2016 -0500\n\n    cgroup: make css_tryget_online_from_dir() also recognize cgroup2 fs\n\n    The function currently returns -EBADF for a directory on the default\n    hierarchy.  Make it also recognize cgroup2_fs_type.  This will be used\n    for perf_event cgroup2 support.\n\n    Signed-off-by: Tejun Heo \u003ctj@kernel.org\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 32ed0cca13bfabfd89a912fd0713d244d919c0fa\nAuthor: Tejun Heo \u003ctj@kernel.org\u003e\nDate:   Tue Feb 23 10:00:50 2016 -0500\n\n    cgroup: s/cgrp_dfl_root_/cgrp_dfl_/\n\n    These var names are unnecessarily unwiedly and another similar\n    variable will be added.  Let\u0027s shorten them.\n\n    Signed-off-by: Tejun Heo \u003ctj@kernel.org\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 0a98d3462642e0a257a6566a1398190c96ec97aa\nAuthor: Tejun Heo \u003ctj@kernel.org\u003e\nDate:   Mon Feb 22 22:25:47 2016 -0500\n\n    cgroup: make cgroup subsystem masks u16\n\n    After the recent do_each_subsys_mask() conversion, there\u0027s no reason\n    to use ulong for subsystem masks.  We\u0027ll be adding more subsystem\n    masks to persistent data structures, let\u0027s reduce its size to u16\n    which should be enough for now and the foreseeable future.\n\n    This doesn\u0027t create any noticeable behavior differences.\n\n    v2: Johannes spotted that the initial patch missed cgroup_no_v1_mask.\n        Converted.\n\n    Signed-off-by: Tejun Heo \u003ctj@kernel.org\u003e\n    Acked-by: Johannes Weiner \u003channes@cmpxchg.org\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 0b90c5a2978b31d2b28114581c5667a7cdd4948a\nAuthor: Tejun Heo \u003ctj@kernel.org\u003e\nDate:   Mon Feb 22 22:25:46 2016 -0500\n\n    cgroup: use do_each_subsys_mask() where applicable\n\n    There are several places in cgroup_subtree_control_write() which can\n    use do_each_subsys_mask() instead of manual mask testing.  Use it.\n\n    No functional changes.\n\n    Signed-off-by: Tejun Heo \u003ctj@kernel.org\u003e\n    Acked-by: Johannes Weiner \u003channes@cmpxchg.org\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 346d2aba56ce19e4a602f572263f33198353eb20\nAuthor: Tejun Heo \u003ctj@kernel.org\u003e\nDate:   Mon Feb 22 22:25:46 2016 -0500\n\n    cgroup: convert for_each_subsys_which() to do-while style\n\n    for_each_subsys_which() allows iterating subsystems specified in a\n    subsystem bitmask; unfortunately, it requires the mask to be an\n    unsigned long l-value which can be inconvenient and makes it awkward\n    to use a smaller type for subsystem masks.\n\n    This patch converts for_each_subsy_which() to do-while style which\n    allows it to drop the l-value requirement.  The new iterator is named\n    do_each_subsys_mask() / while_each_subsys_mask().\n\n    Signed-off-by: Tejun Heo \u003ctj@kernel.org\u003e\n    Cc: Aleksa Sarai \u003ccyphar@cyphar.com\u003e\n    Acked-by: Johannes Weiner \u003channes@cmpxchg.org\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit a4b5f7dd33516068b46815292640efd74d477ee4\nAuthor: Tejun Heo \u003ctj@kernel.org\u003e\nDate:   Mon Feb 22 22:25:46 2016 -0500\n\n    cgroup: s/child_subsys_mask/subtree_ss_mask/\n\n    For consistency with cgroup-\u003esubtree_control.\n\n    * cgroup-\u003echild_subsys_mask -\u003e cgroup-\u003esubtree_ss_mask\n    * cgroup_calc_child_subsys_mask() -\u003e cgroup_calc_subtree_ss_mask()\n    * cgroup_refresh_child_subsys_mask() -\u003e cgroup_refresh_subtree_ss_mask()\n\n    No functional changes.\n\n    Signed-off-by: Tejun Heo \u003ctj@kernel.org\u003e\n    Acked-by: Johannes Weiner \u003channes@cmpxchg.org\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 0317250e21ee50cb18dd1f86efddfa38f2953f9d\nAuthor: Tejun Heo \u003ctj@kernel.org\u003e\nDate:   Mon Feb 22 22:25:46 2016 -0500\n\n    Revert \"cgroup: add cgroup_subsys-\u003ecss_e_css_changed()\"\n\n    This reverts commit 56c807ba4e91f0980567b6a69de239677879b17f.\n\n    cgroup_subsys-\u003ecss_e_css_changed() was supposed to be used by cgroup\n    writeback support; however, the change to per-inode cgroup association\n    made it unnecessary and the callback doesn\u0027t have any user.  Remove\n    it.\n\n    Signed-off-by: Tejun Heo \u003ctj@kernel.org\u003e\n    Acked-by: Johannes Weiner \u003channes@cmpxchg.org\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 1b0104f0af0922a6de76ee0eaa2af1ef63ced0de\nAuthor: Tejun Heo \u003ctj@kernel.org\u003e\nDate:   Mon Feb 22 22:25:45 2016 -0500\n\n    cgroup: fix error return value of cgroup_addrm_files()\n\n    cgroup_addrm_files() incorrectly returned 0 after add failure.  Fix\n    it.\n\n    Signed-off-by: Tejun Heo \u003ctj@kernel.org\u003e\n    Acked-by: Johannes Weiner \u003channes@cmpxchg.org\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 959c3f17a0eedc34992cde0c4c7e3d3225d6d8b4\nAuthor: Tejun Heo \u003ctj@kernel.org\u003e\nDate:   Thu Feb 18 11:44:24 2016 -0500\n\n    cgroup: fix alloc_cgroup_ns() error handling in copy_cgroup_ns()\n\n    alloc_cgroup_ns() returns an ERR_PTR value on error but\n    copy_cgroup_ns() was checking for NULL for error.  Fix it.\n\n    Signed-off-by: Tejun Heo \u003ctj@kernel.org\u003e\n    Reported-by: Dan Carpenter \u003cdan.carpenter@oracle.com\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 6a2f54048873dfc28666c7c8c2e006f774eba02b\nAuthor: Serge Hallyn \u003cserge.hallyn@ubuntu.com\u003e\nDate:   Fri Jan 29 02:54:11 2016 -0600\n\n    Add FS_USERNS_FLAG to cgroup fs\n\n    allowing root in a non-init user namespace to mount it.  This should\n    now be safe, because\n\n    1. non-init-root cannot mount a previously unbound subsystem\n    2. the task doing the mount must be privileged with respect to the\n       user namespace owning the cgroup namespace\n    3. the mounted subsystem will have its current cgroup as the root dentry.\n       the permissions will be unchanged, so tasks will receive no new\n       privilege over the cgroups which they did not have on the original\n       mounts.\n\n    Signed-off-by: Serge Hallyn \u003cserge.hallyn@canonical.com\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 48a66647cab717b65b80e821475a3b89295db092\nAuthor: Serge Hallyn \u003cserge.hallyn@ubuntu.com\u003e\nDate:   Fri Jan 29 02:54:09 2016 -0600\n\n    cgroup: mount cgroupns-root when inside non-init cgroupns\n\n    This patch enables cgroup mounting inside userns when a process\n    as appropriate privileges. The cgroup filesystem mounted is\n    rooted at the cgroupns-root. Thus, in a container-setup, only\n    the hierarchy under the cgroupns-root is exposed inside the container.\n    This allows container management tools to run inside the containers\n    without depending on any global state.\n\n    Signed-off-by: Serge Hallyn \u003cserge.hallyn@canonical.com\u003e\n    Signed-off-by: Tejun Heo \u003ctj@kernel.org\u003e\n    Signed-off-by: Tejun Heo \u003ctj@kernel.org\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit d14c56b424aeac181246e5a2b7303b538d919bca\nAuthor: Aditya Kali \u003cadityakali@google.com\u003e\nDate:   Fri Jan 29 02:54:07 2016 -0600\n\n    cgroup: cgroup namespace setns support\n\n    setns on a cgroup namespace is allowed only if\n    task has CAP_SYS_ADMIN in its current user-namespace and\n    over the user-namespace associated with target cgroupns.\n    No implicit cgroup changes happen with attaching to another\n    cgroupns. It is expected that the somone moves the attaching\n    process under the target cgroupns-root.\n\n    Signed-off-by: Aditya Kali \u003cadityakali@google.com\u003e\n    Signed-off-by: Serge E. Hallyn \u003cserge.hallyn@canonical.com\u003e\n    Signed-off-by: Tejun Heo \u003ctj@kernel.org\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit f1c91cd3381c6f0100f6cca723e760373e74ee6e\nAuthor: Aditya Kali \u003cadityakali@google.com\u003e\nDate:   Fri Jan 29 02:54:06 2016 -0600\n\n    cgroup: introduce cgroup namespaces\n\n    Introduce the ability to create new cgroup namespace. The newly created\n    cgroup namespace remembers the cgroup of the process at the point\n    of creation of the cgroup namespace (referred as cgroupns-root).\n    The main purpose of cgroup namespace is to virtualize the contents\n    of /proc/self/cgroup file. Processes inside a cgroup namespace\n    are only able to see paths relative to their namespace root\n    (unless they are moved outside of their cgroupns-root, at which point\n     they will see a relative path from their cgroupns-root).\n    For a correctly setup container this enables container-tools\n    (like libcontainer, lxc, lmctfy, etc.) to create completely virtualized\n    containers without leaking system level cgroup hierarchy to the task.\n    This patch only implements the \u0027unshare\u0027 part of the cgroupns.\n\n    Signed-off-by: Aditya Kali \u003cadityakali@google.com\u003e\n    Signed-off-by: Serge Hallyn \u003cserge.hallyn@canonical.com\u003e\n    Signed-off-by: Tejun Heo \u003ctj@kernel.org\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n    Change-Id: I41c6f048567d7ab086467c6a230780c1858b315d\n\ncommit c09890431c1e95bb6a103d53dcb0784a73b68b04\nAuthor: Johannes Weiner \u003channes@cmpxchg.org\u003e\nDate:   Thu Feb 11 13:34:49 2016 -0500\n\n    cgroup: provide cgroup_nov1\u003d to disable controllers in v1 mounts\n\n    Testing cgroup2 can be painful with system software automatically\n    mounting and populating all cgroup controllers in v1 mode. Sometimes\n    they can be unmounted from rc.local, sometimes even that is too late.\n\n    Provide a commandline option to disable certain controllers in v1\n    mounts, so that they remain available for cgroup2 mounts.\n\n    Example use:\n\n    cgroup_no_v1\u003dmemory,cpu\n    cgroup_no_v1\u003dall\n\n    Disabling will be confirmed at boot-time as such:\n\n    [    0.013770] Disabling cpu control group subsystem in v1 mounts\n    [    0.016004] Disabling memory control group subsystem in v1 mounts\n\n    Signed-off-by: Johannes Weiner \u003channes@cmpxchg.org\u003e\n    Signed-off-by: Tejun Heo \u003ctj@kernel.org\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 08730efa99ba420805acf2dbaa5aa4f2be5238bd\nAuthor: Tejun Heo \u003ctj@kernel.org\u003e\nDate:   Tue Dec 29 14:53:56 2015 -0500\n\n    cgroup: demote subsystem init messages to KERN_DEBUG\n\n    These are noisy during boot and not all that interesting.\n\n    Signed-off-by: Tejun Heo \u003ctj@kernel.org\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit f6463c52a5107937c1b383bca79a2143f533b225\nAuthor: Rami Rosen \u003crami.rosen@intel.com\u003e\nDate:   Sat Jan 9 23:33:06 2016 +0200\n\n    cgroup: fix a typo.\n\n    This patch fixes a typo in a comment in cgroup.c.\n\n    Signed-off-by: Rami Rosen \u003crami.rosen@intel.com\u003e\n    Signed-off-by: Tejun Heo \u003ctj@kernel.org\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 0d1d0895cd450944c5b82914fcc35808b2224915\nAuthor: Tejun Heo \u003ctj@kernel.org\u003e\nDate:   Mon Dec 14 11:24:06 2015 -0500\n\n    net, cgroup: cgroup_sk_updat_lock was missing initializer\n\n    bd1060a1d671 (\"sock, cgroup: add sock-\u003esk_cgroup\") added global\n    spinlock cgroup_sk_update_lock but erroneously skipped initializer\n    leading to uninitialized spinlock warning.  Fix it by using\n    DEFINE_SPINLOCK().\n\n    Signed-off-by: Tejun Heo \u003ctj@kernel.org\u003e\n    Reported-by: Dexuan Cui \u003cdecui@microsoft.com\u003e\n    Fixes: bd1060a1d671 (\"sock, cgroup: add sock-\u003esk_cgroup\")\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 4daebc81aa70048462f8290069926d378651256e\nAuthor: Tejun Heo \u003ctj@kernel.org\u003e\nDate:   Mon Dec 7 17:38:53 2015 -0500\n\n    sock, cgroup: add sock-\u003esk_cgroup\n\n    In cgroup v1, dealing with cgroup membership was difficult because the\n    number of membership associations was unbound.  As a result, cgroup v1\n    grew several controllers whose primary purpose is either tagging\n    membership or pull in configuration knobs from other subsystems so\n    that cgroup membership test can be avoided.\n\n    net_cls and net_prio controllers are examples of the latter.  They\n    allow configuring network-specific attributes from cgroup side so that\n    network subsystem can avoid testing cgroup membership; unfortunately,\n    these are not only cumbersome but also problematic.\n\n    Both net_cls and net_prio aren\u0027t properly hierarchical.  Both inherit\n    configuration from the parent on creation but there\u0027s no interaction\n    afterwards.  An ancestor doesn\u0027t restrict the behavior in its subtree\n    in anyway and configuration changes aren\u0027t propagated downwards.\n    Especially when combined with cgroup delegation, this is problematic\n    because delegatees can mess up whatever network configuration\n    implemented at the system level.  net_prio would allow the delegatees\n    to set whatever priority value regardless of CAP_NET_ADMIN and net_cls\n    the same for classid.\n\n    While it is possible to solve these issues from controller side by\n    implementing hierarchical allowable ranges in both controllers, it\n    would involve quite a bit of complexity in the controllers and further\n    obfuscate network configuration as it becomes even more difficult to\n    tell what\u0027s actually being configured looking from the network side.\n    While not much can be done for v1 at this point, as membership\n    handling is sane on cgroup v2, it\u0027d be better to make cgroup matching\n    behave like other network matches and classifiers than introducing\n    further complications.\n\n    In preparation, this patch updates sock-\u003esk_cgrp_data handling so that\n    it points to the v2 cgroup that sock was created in until either\n    net_prio or net_cls is used.  Once either of the two is used,\n    sock-\u003esk_cgrp_data reverts to its previous role of carrying prioidx\n    and classid.  This is to avoid adding yet another cgroup related field\n    to struct sock.\n\n    As the mode switching can happen at most once per boot, the switching\n    mechanism is aimed at lowering hot path overhead.  It may leak a\n    finite, likely small, number of cgroup refs and report spurious\n    prioidx or classid on switching; however, dynamic updates of prioidx\n    and classid have always been racy and lossy - socks between creation\n    and fd installation are never updated, config changes don\u0027t update\n    existing sockets at all, and prioidx may index with dead and recycled\n    cgroup IDs.  Non-critical inaccuracies from small race windows won\u0027t\n    make any noticeable difference.\n\n    This patch doesn\u0027t make use of the pointer yet.  The following patch\n    will implement netfilter match for cgroup2 membership.\n\n    v2: Use sock_cgroup_data to avoid inflating struct sock w/ another\n        cgroup specific field.\n\n    v3: Add comments explaining why sock_data_prioidx() and\n        sock_data_classid() use different fallback values.\n\n    Signed-off-by: Tejun Heo \u003ctj@kernel.org\u003e\n    Cc: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Cc: Daniel Wagner \u003cdaniel.wagner@bmw-carit.de\u003e\n    CC: Neil Horman \u003cnhorman@tuxdriver.com\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit bcc4436d162b13051f4ef73b0386bb73dff412ef\nAuthor: Tejun Heo \u003ctj@kernel.org\u003e\nDate:   Mon Dec 7 17:38:52 2015 -0500\n\n    net: wrap sock-\u003esk_cgrp_prioidx and -\u003esk_classid inside a struct\n\n    Introduce sock-\u003esk_cgrp_data which is a struct sock_cgroup_data.\n    -\u003esk_cgroup_prioidx and -\u003esk_classid are moved into it.  The struct\n    and its accessors are defined in cgroup-defs.h.  This is to prepare\n    for overloading the fields with a cgroup pointer.\n\n    This patch mostly performs equivalent conversions but the followings\n    are noteworthy.\n\n    * Equality test before updating classid is removed from\n      sock_update_classid().  This shouldn\u0027t make any noticeable\n      difference and a similar test will be implemented on the helper side\n      later.\n\n    * sock_update_netprioidx() now takes struct sock_cgroup_data and can\n      be moved to netprio_cgroup.h without causing include dependency\n      loop.  Moved.\n\n    * The dummy version of sock_update_netprioidx() converted to a static\n      inline function while at it.\n\n    Signed-off-by: Tejun Heo \u003ctj@kernel.org\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 63b1c5c5c5a3ccf4a190577b5288ae6a581a186a\nAuthor: Tejun Heo \u003ctj@kernel.org\u003e\nDate:   Mon Dec 7 17:38:51 2015 -0500\n\n    netprio_cgroup: limit the maximum css-\u003eid to USHRT_MAX\n\n    netprio builds per-netdev contiguous priomap array which is indexed by\n    css-\u003eid.  The array is allocated using kzalloc() effectively limiting\n    the maximum ID supported to some thousand range.  This patch caps the\n    maximum supported css-\u003eid to USHRT_MAX which should be way above what\n    is actually useable.\n\n    This allows reducing sock-\u003esk_cgrp_prioidx to u16 from u32.  The freed\n    up part will be used to overload the cgroup related fields.\n    sock-\u003esk_cgrp_prioidx\u0027s position is swapped with sk_mark so that the\n    two cgroup related fields are adjacent.\n\n    Signed-off-by: Tejun Heo \u003ctj@kernel.org\u003e\n    Acked-by: Daniel Wagner \u003cdaniel.wagner@bmw-carit.de\u003e\n    Cc: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    CC: Neil Horman \u003cnhorman@tuxdriver.com\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit e98dcefa5b11c7d168adf8cb601a62a3d1e66450\nAuthor: Oleg Nesterov \u003coleg@redhat.com\u003e\nDate:   Thu Dec 3 10:24:08 2015 -0500\n\n    cgroup: kill cgrp_ss_priv[CGROUP_CANFORK_COUNT] and friends\n\n    Now that nobody use the \"priv\" arg passed to can_fork/cancel_fork/fork we can\n    kill CGROUP_CANFORK_COUNT/SUBSYS_TAG/etc and cgrp_ss_priv[] in copy_process().\n\n    Signed-off-by: Oleg Nesterov \u003coleg@redhat.com\u003e\n    Signed-off-by: Tejun Heo \u003ctj@kernel.org\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n    Change-Id: I153eb067c3378de42ccfd4bf114763032bb260ec\n\ncommit 283d51037e28a568451bd41fa5e90a3ce478cb0b\nAuthor: Tejun Heo \u003ctj@kernel.org\u003e\nDate:   Fri Nov 20 15:55:52 2015 -0500\n\n    cgroup: implement cgroup_get_from_path() and expose cgroup_put()\n\n    Implement cgroup_get_from_path() using kernfs_walk_and_get() which\n    obtains a default hierarchy cgroup from its path.  This will be used\n    to allow cgroup path based matching from outside cgroup proper -\n    e.g. networking and perf.\n\n    v2: Add EXPORT_SYMBOL_GPL(cgroup_get_from_path).\n\n    Signed-off-by: Tejun Heo \u003ctj@kernel.org\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit e397062e7a210236c7d184e029b587b81f57f5d2\nAuthor: Tejun Heo \u003ctj@kernel.org\u003e\nDate:   Fri Nov 20 15:55:52 2015 -0500\n\n    cgroup: record ancestor IDs and reimplement cgroup_is_descendant() using it\n\n    cgroup_is_descendant() currently walks up the hierarchy and compares\n    each ancestor to the cgroup in question.  While enough for cgroup core\n    usages, this can\u0027t be used in hot paths to test cgroup membership.\n    This patch adds cgroup-\u003eancestor_ids[] which records the IDs of all\n    ancestors including self and cgroup-\u003elevel for the nesting level.\n\n    This allows testing whether a given cgroup is a descendant of another\n    in three finite steps - testing whether the two belong to the same\n    hierarchy, whether the descendant candidate is at the same or a higher\n    level than the ancestor and comparing the recorded ancestor_id at the\n    matching level.  cgroup_is_descendant() is accordingly reimplmented\n    and made inline.\n\n    Signed-off-by: Tejun Heo \u003ctj@kernel.org\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit f4cdf9e7884cb2e332c9745116cbbd331e5c95c0\nAuthor: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\nDate:   Mon May 8 00:04:09 2017 +0200\n\n    bpf: don\u0027t let ldimm64 leak map addresses on unprivileged\n\n    [ Upstream commit 0d0e57697f162da4aa218b5feafe614fb666db07 ]\n\n    The patch fixes two things at once:\n\n    1) It checks the env-\u003eallow_ptr_leaks and only prints the map address to\n       the log if we have the privileges to do so, otherwise it just dumps 0\n       as we would when kptr_restrict is enabled on %pK. Given the latter is\n       off by default and not every distro sets it, I don\u0027t want to rely on\n       this, hence the 0 by default for unprivileged.\n\n    2) Printing of ldimm64 in the verifier log is currently broken in that\n       we don\u0027t print the full immediate, but only the 32 bit part of the\n       first insn part for ldimm64. Thus, fix this up as well; it\u0027s okay to\n       access, since we verified all ldimm64 earlier already (including just\n       constants) through replace_map_fd_with_map_ptr().\n\n    Fixes: 1be7f75d1668 (\"bpf: enable non-root eBPF programs\")\n    Fixes: cbd357008604 (\"bpf: verifier (add ability to receive verification log)\")\n    Reported-by: Jann Horn \u003cjannh@google.com\u003e\n    Signed-off-by: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Acked-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Greg Kroah-Hartman \u003cgregkh@linuxfoundation.org\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 94f835b9b83d3da72ee4773fe2014b1cf619dbe4\nAuthor: Yonghong Song \u003cyhs@fb.com\u003e\nDate:   Sat Apr 29 22:52:42 2017 -0700\n\n    bpf: enhance verifier to understand stack pointer arithmetic\n\n    [ Upstream commit 332270fdc8b6fba07d059a9ad44df9e1a2ad4529 ]\n\n    llvm 4.0 and above generates the code like below:\n    ....\n    440: (b7) r1 \u003d 15\n    441: (05) goto pc+73\n    515: (79) r6 \u003d *(u64 *)(r10 -152)\n    516: (bf) r7 \u003d r10\n    517: (07) r7 +\u003d -112\n    518: (bf) r2 \u003d r7\n    519: (0f) r2 +\u003d r1\n    520: (71) r1 \u003d *(u8 *)(r8 +0)\n    521: (73) *(u8 *)(r2 +45) \u003d r1\n    ....\n    and the verifier complains \"R2 invalid mem access \u0027inv\u0027\" for insn #521.\n    This is because verifier marks register r2 as unknown value after #519\n    where r2 is a stack pointer and r1 holds a constant value.\n\n    Teach verifier to recognize \"stack_ptr + imm\" and\n    \"stack_ptr + reg with const val\" as valid stack_ptr with new offset.\n\n    Signed-off-by: Yonghong Song \u003cyhs@fb.com\u003e\n    Acked-by: Martin KaFai Lau \u003ckafai@fb.com\u003e\n    Acked-by: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Signed-off-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Greg Kroah-Hartman \u003cgregkh@linuxfoundation.org\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 4252230cce4e92914affecd9cd2c50300db9fd8b\nAuthor: Alexei Starovoitov \u003cast@fb.com\u003e\nDate:   Fri Mar 24 15:57:33 2017 -0700\n\n    bpf: improve verifier packet range checks\n\n    [ Upstream commit b1977682a3858b5584ffea7cfb7bd863f68db18d ]\n\n    llvm can optimize the \u0027if (ptr \u003e data_end)\u0027 checks to be in the order\n    slightly different than the original C code which will confuse verifier.\n    Like:\n    if (ptr + 16 \u003e data_end)\n      return TC_ACT_SHOT;\n    // may be followed by\n    if (ptr + 14 \u003e data_end)\n      return TC_ACT_SHOT;\n    while llvm can see that \u0027ptr\u0027 is valid for all 16 bytes,\n    the verifier could not.\n    Fix verifier logic to account for such case and add a test.\n\n    Reported-by: Huapeng Zhou \u003chzhou@fb.com\u003e\n    Fixes: 969bf05eb3ce (\"bpf: direct packet access\")\n    Signed-off-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Acked-by: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Acked-by: Martin KaFai Lau \u003ckafai@fb.com\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Greg Kroah-Hartman \u003cgregkh@linuxfoundation.org\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit ef3e919d24610f2a2be4849d84af8f24b3d4efb7\nAuthor: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\nDate:   Sun Dec 18 01:52:59 2016 +0100\n\n    bpf: fix mark_reg_unknown_value for spilled regs on map value marking\n\n    [ Upstream commit 6760bf2ddde8ad64f8205a651223a93de3a35494 ]\n\n    Martin reported a verifier issue that hit the BUG_ON() for his\n    test case in the mark_reg_unknown_value() function:\n\n      [  202.861380] kernel BUG at kernel/bpf/verifier.c:467!\n      [...]\n      [  203.291109] Call Trace:\n      [  203.296501]  [\u003cffffffff811364d5\u003e] mark_map_reg+0x45/0x50\n      [  203.308225]  [\u003cffffffff81136558\u003e] mark_map_regs+0x78/0x90\n      [  203.320140]  [\u003cffffffff8113938d\u003e] do_check+0x226d/0x2c90\n      [  203.331865]  [\u003cffffffff8113a6ab\u003e] bpf_check+0x48b/0x780\n      [  203.343403]  [\u003cffffffff81134c8e\u003e] bpf_prog_load+0x27e/0x440\n      [  203.355705]  [\u003cffffffff8118a38f\u003e] ? handle_mm_fault+0x11af/0x1230\n      [  203.369158]  [\u003cffffffff812d8188\u003e] ? security_capable+0x48/0x60\n      [  203.382035]  [\u003cffffffff811351a4\u003e] SyS_bpf+0x124/0x960\n      [  203.393185]  [\u003cffffffff810515f6\u003e] ? __do_page_fault+0x276/0x490\n      [  203.406258]  [\u003cffffffff816db320\u003e] entry_SYSCALL_64_fastpath+0x13/0x94\n\n    This issue got uncovered after the fix in a08dd0da5307 (\"bpf: fix\n    regression on verifier pruning wrt map lookups\"). The reason why it\n    wasn\u0027t noticed before was, because as mentioned in a08dd0da5307,\n    mark_map_regs() was doing the id matching incorrectly based on the\n    uncached regs[regno].id. So, in the first loop, we walked all regs\n    and as soon as we found regno \u003d\u003d i, then this reg\u0027s id was cleared\n    when calling mark_reg_unknown_value() thus that every subsequent\n    register was probed against id of 0 (which, in combination with the\n    PTR_TO_MAP_VALUE_OR_NULL type is an invalid condition that no other\n    register state can hold), and therefore wasn\u0027t type transitioned such\n    as in the spilled register case for the second loop.\n\n    Now since that got fixed, it turned out that 57a09bf0a416 (\"bpf:\n    Detect identical PTR_TO_MAP_VALUE_OR_NULL registers\") used\n    mark_reg_unknown_value() incorrectly for the spilled regs, and thus\n    hitting the BUG_ON() in some cases due to regno \u003e\u003d MAX_BPF_REG.\n\n    Although spilled regs have the same type as the non-spilled regs\n    for the verifier state, that is, struct bpf_reg_state, they are\n    semantically different from the non-spilled regs. In other words,\n    there can be up to 64 (MAX_BPF_STACK / BPF_REG_SIZE) spilled regs\n    in the stack, for example, register R\u003cx\u003e could have been spilled by\n    the program to stack location X, Y, Z, and in mark_map_regs() we\n    need to scan these stack slots of type STACK_SPILL for potential\n    registers that we have to transition from PTR_TO_MAP_VALUE_OR_NULL.\n    Therefore, depending on the location, the spilled_regs regno can\n    be a lot higher than just MAX_BPF_REG\u0027s value since we operate on\n    stack instead. The reset in mark_reg_unknown_value() itself is\n    just fine, only that the BUG_ON() was inappropriate for this. Fix\n    it by making a __mark_reg_unknown_value() version that can be\n    called from mark_map_reg() generically; we know for the non-spilled\n    case that the regno is always \u003c MAX_BPF_REG anyway.\n\n    Fixes: 57a09bf0a416 (\"bpf: Detect identical PTR_TO_MAP_VALUE_OR_NULL registers\")\n    Reported-by: Martin KaFai Lau \u003ckafai@fb.com\u003e\n    Signed-off-by: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Acked-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Greg Kroah-Hartman \u003cgregkh@linuxfoundation.org\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 53024f8bfa94807c76a29820f2fcd0f015311f65\nAuthor: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\nDate:   Thu Dec 15 01:30:06 2016 +0100\n\n    bpf: fix regression on verifier pruning wrt map lookups\n\n    [ Upstream commit a08dd0da5307ba01295c8383923e51e7997c3576 ]\n\n    Commit 57a09bf0a416 (\"bpf: Detect identical PTR_TO_MAP_VALUE_OR_NULL\n    registers\") introduced a regression where existing programs stopped\n    loading due to reaching the verifier\u0027s maximum complexity limit,\n    whereas prior to this commit they were loading just fine; the affected\n    program has roughly 2k instructions.\n\n    What was found is that state pruning couldn\u0027t be performed effectively\n    anymore due to mismatches of the verifier\u0027s register state, in particular\n    in the id tracking. It doesn\u0027t mean that 57a09bf0a416 is incorrect per\n    se, but rather that verifier needs to perform a lot more work for the\n    same program with regards to involved map lookups.\n\n    Since commit 57a09bf0a416 is only about tracking registers with type\n    PTR_TO_MAP_VALUE_OR_NULL, the id is only needed to follow registers\n    until they are promoted through pattern matching with a NULL check to\n    either PTR_TO_MAP_VALUE or UNKNOWN_VALUE type. After that point, the\n    id becomes irrelevant for the transitioned types.\n\n    For UNKNOWN_VALUE, id is already reset to 0 via mark_reg_unknown_value(),\n    but not so for PTR_TO_MAP_VALUE where id is becoming stale. It\u0027s even\n    transferred further into other types that don\u0027t make use of it. Among\n    others, one example is where UNKNOWN_VALUE is set on function call\n    return with RET_INTEGER return type.\n\n    states_equal() will then fall through the memcmp() on register state;\n    note that the second memcmp() uses offsetofend(), so the id is part of\n    that since d2a4dd37f6b4 (\"bpf: fix state equivalence\"). But the bisect\n    pointed already to 57a09bf0a416, where we really reach beyond complexity\n    limit. What I found was that states_equal() often failed in this\n    case due to id mismatches in spilled regs with registers in type\n    PTR_TO_MAP_VALUE. Unlike non-spilled regs, spilled regs just perform\n    a memcmp() on their reg state and don\u0027t have any other optimizations\n    in place, therefore also id was relevant in this case for making a\n    pruning decision.\n\n    We can safely reset id to 0 as well when converting to PTR_TO_MAP_VALUE.\n    For the affected program, it resulted in a ~17 fold reduction of\n    complexity and let the program load fine again. Selftest suite also\n    runs fine. The only other place where env-\u003eid_gen is used currently is\n    through direct packet access, but for these cases id is long living, thus\n    a different scenario.\n\n    Also, the current logic in mark_map_regs() is not fully correct when\n    marking NULL branch with UNKNOWN_VALUE. We need to cache the destination\n    reg\u0027s id in any case. Otherwise, once we marked that reg as UNKNOWN_VALUE,\n    it\u0027s id is reset and any subsequent registers that hold the original id\n    and are of type PTR_TO_MAP_VALUE_OR_NULL won\u0027t be marked UNKNOWN_VALUE\n    anymore, since mark_map_reg() reuses the uncached regs[regno].id that\n    was just overridden. Note, we don\u0027t need to cache it outside of\n    mark_map_regs(), since it\u0027s called once on this_branch and the other\n    time on other_branch, which are both two independent verifier states.\n    A test case for this is added here, too.\n\n    Fixes: 57a09bf0a416 (\"bpf: Detect identical PTR_TO_MAP_VALUE_OR_NULL registers\")\n    Signed-off-by: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Acked-by: Thomas Graf \u003ctgraf@suug.ch\u003e\n    Acked-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Greg Kroah-Hartman \u003cgregkh@linuxfoundation.org\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 4f4d7ac9ac2783764c9ec16153eab792abec40a0\nAuthor: Alexei Starovoitov \u003cast@fb.com\u003e\nDate:   Wed Dec 7 10:57:59 2016 -0800\n\n    bpf: fix state equivalence\n\n    [ Upstream commit d2a4dd37f6b41fbcad76efbf63124eb3126c66fe ]\n\n    Commmits 57a09bf0a416 (\"bpf: Detect identical PTR_TO_MAP_VALUE_OR_NULL registers\")\n    and 484611357c19 (\"bpf: allow access into map value arrays\") by themselves\n    are correct, but in combination they make state equivalence ignore \u0027id\u0027 field\n    of the register state which can lead to accepting invalid program.\n\n    Fixes: 57a09bf0a416 (\"bpf: Detect identical PTR_TO_MAP_VALUE_OR_NULL registers\")\n    Fixes: 484611357c19 (\"bpf: allow access into map value arrays\")\n    Signed-off-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Acked-by: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Acked-by: Thomas Graf \u003ctgraf@suug.ch\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Greg Kroah-Hartman \u003cgregkh@linuxfoundation.org\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit ad0e8f5401a20f2fe5da005f1aa9573a0e001f84\nAuthor: Thomas Graf \u003ctgraf@suug.ch\u003e\nDate:   Tue Oct 18 19:51:19 2016 +0200\n\n    bpf: Detect identical PTR_TO_MAP_VALUE_OR_NULL registers\n\n    [ Upstream commit 57a09bf0a416700676e77102c28f9cfcb48267e0 ]\n\n    A BPF program is required to check the return register of a\n    map_elem_lookup() call before accessing memory. The verifier keeps\n    track of this by converting the type of the result register from\n    PTR_TO_MAP_VALUE_OR_NULL to PTR_TO_MAP_VALUE after a conditional\n    jump ensures safety. This check is currently exclusively performed\n    for the result register 0.\n\n    In the event the compiler reorders instructions, BPF_MOV64_REG\n    instructions may be moved before the conditional jump which causes\n    them to keep their type PTR_TO_MAP_VALUE_OR_NULL to which the\n    verifier objects when the register is accessed:\n\n    0: (b7) r1 \u003d 10\n    1: (7b) *(u64 *)(r10 -8) \u003d r1\n    2: (bf) r2 \u003d r10\n    3: (07) r2 +\u003d -8\n    4: (18) r1 \u003d 0x59c00000\n    6: (85) call 1\n    7: (bf) r4 \u003d r0\n    8: (15) if r0 \u003d\u003d 0x0 goto pc+1\n     R0\u003dmap_value(ks\u003d8,vs\u003d8) R4\u003dmap_value_or_null(ks\u003d8,vs\u003d8) R10\u003dfp\n    9: (7a) *(u64 *)(r4 +0) \u003d 0\n    R4 invalid mem access \u0027map_value_or_null\u0027\n\n    This commit extends the verifier to keep track of all identical\n    PTR_TO_MAP_VALUE_OR_NULL registers after a map_elem_lookup() by\n    assigning them an ID and then marking them all when the conditional\n    jump is observed.\n\n    Signed-off-by: Thomas Graf \u003ctgraf@suug.ch\u003e\n    Reviewed-by: Josef Bacik \u003cjbacik@fb.com\u003e\n    Acked-by: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Acked-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Greg Kroah-Hartman \u003cgregkh@linuxfoundation.org\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 172f58ef19644fcd8ba36d8ce7c0d83d0acb9d89\nAuthor: Josef Bacik \u003cjbacik@fb.com\u003e\nDate:   Tue Nov 29 12:27:09 2016 -0500\n\n    bpf: fix states equal logic for varlen access\n\n    If we have a branch that looks something like this\n\n    int foo \u003d map-\u003evalue;\n    if (condition) {\n      foo +\u003d blah;\n    } else {\n      foo \u003d bar;\n    }\n    map-\u003earray[foo] \u003d baz;\n\n    We will incorrectly assume that the !condition branch is equal to the condition\n    branch as the register for foo will be UNKNOWN_VALUE in both cases.  We need to\n    adjust this logic to only do this if we didn\u0027t do a varlen access after we\n    processed the !condition branch, otherwise we have different ranges and need to\n    check the other branch as well.\n\n    Fixes: 484611357c19 (\"bpf: allow access into map value arrays\")\n    Reported-by: Jann Horn \u003cjannh@google.com\u003e\n    Signed-off-by: Josef Bacik \u003cjbacik@fb.com\u003e\n    Acked-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Acked-by: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit c630a7b76841d03f46127abc5bf40200946dc5f9\nAuthor: Josef Bacik \u003cjbacik@fb.com\u003e\nDate:   Mon Nov 14 15:45:36 2016 -0500\n\n    bpf: fix range arithmetic for bpf map access\n\n    I made some invalid assumptions with BPF_AND and BPF_MOD that could result in\n    invalid accesses to bpf map entries.  Fix this up by doing a few things\n\n    1) Kill BPF_MOD support.  This doesn\u0027t actually get used by the compiler in real\n    life and just adds extra complexity.\n\n    2) Fix the logic for BPF_AND, don\u0027t allow AND of negative numbers and set the\n    minimum value to 0 for positive AND\u0027s.\n\n    3) Don\u0027t do operations on the ranges if they are set to the limits, as they are\n    by definition undefined, and allowing arithmetic operations on those values\n    could make them appear valid when they really aren\u0027t.\n\n    This fixes the testcase provided by Jann as well as a few other theoretical\n    problems.\n\n    Reported-by: Jann Horn \u003cjannh@google.com\u003e\n    Signed-off-by: Josef Bacik \u003cjbacik@fb.com\u003e\n    Acked-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 7240f24fe3da999d8655484d6da877fadfc6b5bf\nAuthor: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\nDate:   Fri Nov 4 00:01:19 2016 +0100\n\n    bpf: fix htab map destruction when extra reserve is in use\n\n    Commit a6ed3ea65d98 (\"bpf: restore behavior of bpf_map_update_elem\")\n    added an extra per-cpu reserve to the hash table map to restore old\n    behaviour from pre prealloc times. When non-prealloc is in use for a\n    map, then problem is that once a hash table extra element has been\n    linked into the hash-table, and the hash table is destroyed due to\n    refcount dropping to zero, then htab_map_free() -\u003e delete_all_elements()\n    will walk the whole hash table and drop all elements via htab_elem_free().\n    The problem is that the element from the extra reserve is first fed\n    to the wrong backend allocator and eventually freed twice.\n\n    Fixes: a6ed3ea65d98 (\"bpf: restore behavior of bpf_map_update_elem\")\n    Reported-by: Dmitry Vyukov \u003cdvyukov@google.com\u003e\n    Signed-off-by: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Acked-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 0057b23ea73ebd20042d15d75b96d95ab631fabc\nAuthor: Josef Bacik \u003cjbacik@fb.com\u003e\nDate:   Wed Sep 28 10:54:32 2016 -0400\n\n    bpf: allow access into map value arrays\n\n    Suppose you have a map array value that is something like this\n\n    struct foo {\n    \tunsigned iter;\n    \tint array[SOME_CONSTANT];\n    };\n\n    You can easily insert this into an array, but you cannot modify the contents of\n    foo-\u003earray[] after the fact.  This is because we have no way to verify we won\u0027t\n    go off the end of the array at verification time.  This patch provides a start\n    for this work.  We accomplish this by keeping track of a minimum and maximum\n    value a register could be while we\u0027re checking the code.  Then at the time we\n    try to do an access into a MAP_VALUE we verify that the maximum offset into that\n    region is a valid access into that memory region.  So in practice, code such as\n    this\n\n    unsigned index \u003d 0;\n\n    if (foo-\u003eiter \u003e\u003d SOME_CONSTANT)\n    \tfoo-\u003eiter \u003d index;\n    else\n    \tindex \u003d foo-\u003eiter++;\n    foo-\u003earray[index] \u003d bar;\n\n    would be allowed, as we can verify that index will always be between 0 and\n    SOME_CONSTANT-1.  If you wish to use signed values you\u0027ll have to have an extra\n    check to make sure the index isn\u0027t less than 0, or do something like index %\u003d\n    SOME_CONSTANT.\n\n    Signed-off-by: Josef Bacik \u003cjbacik@fb.com\u003e\n    Acked-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 60cf4edb6bc35c4766a3a498c4109534bf8f2ba0\nAuthor: Shaohua Li \u003cshli@fb.com\u003e\nDate:   Tue Sep 27 08:42:41 2016 -0700\n\n    bpf: clean up put_cpu_var usage\n\n    put_cpu_var takes the percpu data, not the data returned from\n    get_cpu_var.\n\n    This doesn\u0027t change the behavior.\n\n    Cc: Tejun Heo \u003ctj@kernel.org\u003e\n    Cc: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Signed-off-by: Shaohua Li \u003cshli@fb.com\u003e\n    Acked-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Acked-by: Tejun Heo \u003ctj@kernel.org\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit f340f5513e703e114493ac95ecebbf47d7881c6c\nAuthor: Mickaël Salaün \u003cmic@digikod.net\u003e\nDate:   Sat Sep 24 20:01:50 2016 +0200\n\n    bpf: Set register type according to is_valid_access()\n\n    This prevent future potential pointer leaks when an unprivileged eBPF\n    program will read a pointer value from its context. Even if\n    is_valid_access() returns a pointer type, the eBPF verifier replace it\n    with UNKNOWN_VALUE. The register value that contains a kernel address is\n    then allowed to leak. Moreover, this fix allows unprivileged eBPF\n    programs to use functions with (legitimate) pointer arguments.\n\n    Not an issue currently since reg_type is only set for PTR_TO_PACKET or\n    PTR_TO_PACKET_END in XDP and TC programs that can only be loaded as\n    privileged. For now, the only unprivileged eBPF program allowed is for\n    socket filtering and all the types from its context are UNKNOWN_VALUE.\n    However, this fix is important for future unprivileged eBPF programs\n    which could use pointers in their context.\n\n    Signed-off-by: Mickaël Salaün \u003cmic@digikod.net\u003e\n    Cc: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Cc: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Acked-by: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Acked-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 3b445710a196d07dc196266bce694fc9ae83004b\nAuthor: Jakub Kicinski \u003cjakub.kicinski@netronome.com\u003e\nDate:   Wed Sep 21 11:43:59 2016 +0100\n\n    bpf: recognize 64bit immediate loads as consts\n\n    When running as parser interpret BPF_LD | BPF_IMM | BPF_DW\n    instructions as loading CONST_IMM with the value stored\n    in imm.  The verifier will continue not recognizing those\n    due to concerns about search space/program complexity\n    increase.\n\n    Signed-off-by: Jakub Kicinski \u003cjakub.kicinski@netronome.com\u003e\n    Acked-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Acked-by: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 9e6fd59da8035b39a770cb1b9508bce785898f38\nAuthor: Jakub Kicinski \u003cjakub.kicinski@netronome.com\u003e\nDate:   Wed Sep 21 11:43:58 2016 +0100\n\n    bpf: enable non-core use of the verfier\n\n    Advanced JIT compilers and translators may want to use\n    eBPF verifier as a base for parsers or to perform custom\n    checks and validations.\n\n    Add ability for external users to invoke the verifier\n    and provide callbacks to be invoked for every intruction\n    checked.  For now only add most basic callback for\n    per-instruction pre-interpretation checks is added.  More\n    advanced users may also like to have per-instruction post\n    callback and state comparison callback.\n\n    Signed-off-by: Jakub Kicinski \u003cjakub.kicinski@netronome.com\u003e\n    Acked-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 2f728ad59eedee6a327dd59d21210f285dd9a0fe\nAuthor: Jakub Kicinski \u003cjakub.kicinski@netronome.com\u003e\nDate:   Wed Sep 21 11:43:57 2016 +0100\n\n    bpf: expose internal verfier structures\n\n    Move verifier\u0027s internal structures to a header file and\n    prefix their names with bpf_ to avoid potential namespace\n    conflicts.  Those structures will soon be used by external\n    analyzers.\n\n    Signed-off-by: Jakub Kicinski \u003cjakub.kicinski@netronome.com\u003e\n    Acked-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Acked-by: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 22e7afab4576f616726444590ee2e5005eb49d3f\nAuthor: Jakub Kicinski \u003cjakub.kicinski@netronome.com\u003e\nDate:   Wed Sep 21 11:43:56 2016 +0100\n\n    bpf: don\u0027t (ab)use instructions to store state\n\n    Storing state in reserved fields of instructions makes\n    it impossible to run verifier on programs already\n    marked as read-only. Allocate and use an array of\n    per-instruction state instead.\n\n    While touching the error path rename and move existing\n    jump target.\n\n    Suggested-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Signed-off-by: Jakub Kicinski \u003cjakub.kicinski@netronome.com\u003e\n    Acked-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Acked-by: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 30d07135dc717446771544bf52ae028e8937d226\nAuthor: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\nDate:   Tue Sep 20 00:26:13 2016 +0200\n\n    bpf: direct packet write and access for helpers for clsact progs\n\n    This work implements direct packet access for helpers and direct packet\n    write in a similar fashion as already available for XDP types via commits\n    4acf6c0b84c9 (\"bpf: enable direct packet data write for xdp progs\") and\n    6841de8b0d03 (\"bpf: allow helpers access the packet directly\"), and as a\n    complementary feature to the already available direct packet read for tc\n    (cls/act) programs.\n\n    For enabling this, we need to introduce two helpers, bpf_skb_pull_data()\n    and bpf_csum_update(). The first is generally needed for both, read and\n    write, because they would otherwise only be limited to the current linear\n    skb head. Usually, when the data_end test fails, programs just bail out,\n    or, in the direct read case, use bpf_skb_load_bytes() as an alternative\n    to overcome this limitation. If such data sits in non-linear parts, we\n    can just pull them in once with the new helper, retest and eventually\n    access them.\n\n    At the same time, this also makes sure the skb is uncloned, which is, of\n    course, a necessary condition for direct write. As this needs to be an\n    invariant for the write part only, the verifier detects writes and adds\n    a prologue that is calling bpf_skb_pull_data() to effectively unclone the\n    skb from the very beginning in case it is indeed cloned. The heuristic\n    makes use of a similar trick that was done in 233577a22089 (\"net: filter:\n    constify detection of pkt_type_offset\"). This comes at zero cost for other\n    programs that do not use the direct write feature. Should a program use\n    this feature only sparsely and has read access for the most parts with,\n    for example, drop return codes, then such write action can be delegated\n    to a tail called program for mitigating this cost of potential uncloning\n    to a late point in time where it would have been paid similarly with the\n    bpf_skb_store_bytes() as well. Advantage of direct write is that the\n    writes are inlined whereas the helper cannot make any length assumptions\n    and thus needs to generate a call to memcpy() also for small sizes, as well\n    as cost of helper call itself with sanity checks are avoided. Plus, when\n    direct read is already used, we don\u0027t need to cache or perform rechecks\n    on the data boundaries (due to verifier invalidating previous checks for\n    helpers that change skb-\u003edata), so more complex programs using rewrites\n    can benefit from switching to direct read plus write.\n\n    For direct packet access to helpers, we save the otherwise needed copy into\n    a temp struct sitting on stack memory when use-case allows. Both facilities\n    are enabled via may_access_direct_pkt_data() in verifier. For now, we limit\n    this to map helpers and csum_diff, and can successively enable other helpers\n    where we find it makes sense. Helpers that definitely cannot be allowed for\n    this are those part of bpf_helper_changes_skb_data() since they can change\n    underlying data, and those that write into memory as this could happen for\n    packet typed args when still cloned. bpf_csum_update() helper accommodates\n    for the fact that we need to fixup checksum_complete when using direct write\n    instead of bpf_skb_store_bytes(), meaning the programs can use available\n    helpers like bpf_csum_diff(), and implement csum_add(), csum_sub(),\n    csum_block_add(), csum_block_sub() equivalents in eBPF together with the\n    new helper. A usage example will be provided for iproute2\u0027s examples/bpf/\n    directory.\n\n    Signed-off-by: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Acked-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 5727cfee236ac7c9c761347a221af1cb0901875e\nAuthor: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\nDate:   Tue Sep 20 00:26:12 2016 +0200\n\n    bpf, verifier: enforce larger zero range for pkt on overloading stack buffs\n\n    Current contract for the following two helper argument types is:\n\n      * ARG_CONST_STACK_SIZE: passed argument pair must be (ptr, \u003e0).\n      * ARG_CONST_STACK_SIZE_OR_ZERO: passed argument pair can be either\n        (NULL, 0) or (ptr, \u003e0).\n\n    With 6841de8b0d03 (\"bpf: allow helpers access the packet directly\"), we can\n    pass also raw packet data to helpers, so depending on the argument type\n    being PTR_TO_PACKET, we now either assert memory via check_packet_access()\n    or check_stack_boundary(). As a result, the tests in check_packet_access()\n    currently allow more than intended with regards to reg-\u003eimm.\n\n    Back in 969bf05eb3ce (\"bpf: direct packet access\"), check_packet_access()\n    was fine to ignore size argument since in check_mem_access() size was\n    bpf_size_to_bytes() derived and prior to the call to check_packet_access()\n    guaranteed to be larger than zero.\n\n    However, for the above two argument types, it currently means, we can have\n    a \u003c\u003d 0 size and thus breaking current guarantees for helpers. Enforce a\n    check for size \u003c\u003d 0 and bail out if so.\n\n    check_stack_boundary() doesn\u0027t have such an issue since it already tests\n    for access_size \u003c\u003d 0 and bails out, resp. access_size \u003d\u003d 0 in case of NULL\n    pointer passed when allowed.\n\n    Fixes: 6841de8b0d03 (\"bpf: allow helpers access the packet directly\")\n    Signed-off-by: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Acked-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 3ecba0c15d9a3b68f5609e8819a8f2326698438b\nAuthor: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\nDate:   Fri Sep 9 02:45:31 2016 +0200\n\n    bpf: add BPF_CALL_x macros for declaring helpers\n\n    This work adds BPF_CALL_\u003cn\u003e() macros and converts all the eBPF helper functions\n    to use them, in a similar fashion like we do with SYSCALL_DEFINE\u003cn\u003e() macros\n    that are used today. Motivation for this is to hide all the register handling\n    and all necessary casts from the user, so that it is done automatically in the\n    background when adding a BPF_CALL_\u003cn\u003e() call.\n\n    This makes current helpers easier to review, eases to write future helpers,\n    avoids getting the casting mess wrong, and allows for extending all helpers at\n    once (f.e. build time checks, etc). It also helps detecting more easily in\n    code reviews that unused registers are not instrumented in the code by accident,\n    breaking compatibility with existing programs.\n\n    BPF_CALL_\u003cn\u003e() internals are quite similar to SYSCALL_DEFINE\u003cn\u003e() ones with some\n    fundamental differences, for example, for generating the actual helper function\n    that carries all u64 regs, we need to fill unused regs, so that we always end up\n    with 5 u64 regs as an argument.\n\n    I reviewed several 0-5 generated BPF_CALL_\u003cn\u003e() variants of the .i results and\n    they look all as expected. No sparse issue spotted. We let this also sit for a\n    few days with Fengguang\u0027s kbuild test robot, and there were no issues seen. On\n    s390, it barked on the \"uses dynamic stack allocation\" notice, which is an old\n    one from bpf_perf_event_output{,_tp}() reappearing here due to the conversion\n    to the call wrapper, just telling that the perf raw record/frag sits on stack\n    (gcc with s390\u0027s -mwarn-dynamicstack), but that\u0027s all. Did various runtime tests\n    and they were fine as well. All eBPF helpers are now converted to use these\n    macros, getting rid of a good chunk of all the raw castings.\n\n    Signed-off-by: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Acked-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 7838443f5d6207bee662afa8d1d4050aa78b5dd2\nAuthor: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\nDate:   Fri Aug 5 00:11:13 2016 +0200\n\n    bpf: fix checksum for vlan push/pop helper\n\n    When having skbs on ingress with CHECKSUM_COMPLETE, tc BPF programs don\u0027t\n    push rcsum of mac header back in and after BPF run back pull out again as\n    opposed to some other subsystems (ovs, for example).\n\n    For cases like q-in-q, meaning when a vlan tag for offloading is already\n    present and we\u0027re about to push another one, then skb_vlan_push() pushes the\n    inner one into the skb, increasing mac header and skb_postpush_rcsum()\u0027ing\n    the 4 bytes vlan header diff. Likewise, for the reverse operation in\n    skb_vlan_pop() for the case where vlan header needs to be pulled out of the\n    skb, we\u0027re decreasing the mac header and skb_postpull_rcsum()\u0027ing the 4 bytes\n    rcsum of the vlan header that was removed.\n\n    However mangling the rcsum here will lead to hw csum failure for BPF case,\n    since we\u0027re pulling or pushing data that was not part of the current rcsum.\n    Changing tc BPF programs in general to push/pull rcsum around BPF_PROG_RUN()\n    is also not really an option since current behaviour is ABI by now, but apart\n    from that would also mean to do quite a bit of useless work in the sense that\n    usually 12 bytes need to be rcsum pushed/pulled also when we don\u0027t need to\n    touch this vlan related corner case. One way to fix it would be to push the\n    necessary rcsum fixup down into vlan helpers that are (mostly) slow-path\n    anyway.\n\n    Fixes: 4e10df9a60d9 (\"bpf: introduce bpf_skb_vlan_push/pop() helpers\")\n    Signed-off-by: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Acked-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 430758126c5690f82ea177e764f5fb3129bdffc7\nAuthor: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\nDate:   Fri Aug 5 00:11:12 2016 +0200\n\n    bpf: fix checksum fixups on bpf_skb_store_bytes\n\n    bpf_skb_store_bytes() invocations above L2 header need BPF_F_RECOMPUTE_CSUM\n    flag for updates, so that CHECKSUM_COMPLETE will be fixed up along the way.\n    Where we ran into an issue with bpf_skb_store_bytes() is when we did a\n    single-byte update on the IPv6 hoplimit despite using BPF_F_RECOMPUTE_CSUM\n    flag; simple ping via ICMPv6 triggered a hw csum failure as a result. The\n    underlying issue has been tracked down to a buffer alignment issue.\n\n    Meaning, that csum_partial() computations via skb_postpull_rcsum() and\n    skb_postpush_rcsum() pair invoked had a wrong result since they operated on\n    an odd address for the hoplimit, while other computations were done on an\n    even address. This mix doesn\u0027t work as-is with skb_postpull_rcsum(),\n    skb_postpush_rcsum() pair as it always expects at least half-word alignment\n    of input buffers, which is normally the case. Thus, instead of these helpers\n    using csum_sub() and (implicitly) csum_add(), we need to use csum_block_sub(),\n    csum_block_add(), respectively. For unaligned offsets, they rotate the sum\n    to align it to a half-word boundary again, otherwise they work the same as\n    csum_sub() and csum_add().\n\n    Adding __skb_postpull_rcsum(), __skb_postpush_rcsum() variants that take the\n    offset as an input and adapting bpf_skb_store_bytes() to them fixes the hw\n    csum failures again. The skb_postpull_rcsum(), skb_postpush_rcsum() helpers\n    use a 0 constant for offset so that the compiler optimizes the offset \u0026 1\n    test away and generates the same code as with csum_sub()/_add().\n\n    Fixes: 608cd71a9c7c (\"tc: bpf: generalize pedit action\")\n    Signed-off-by: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Acked-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit b56fb8850be31217a4cc23c557994a964f3e3640\nAuthor: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\nDate:   Fri Aug 5 00:11:11 2016 +0200\n\n    bpf: also call skb_postpush_rcsum on xmit occasions\n\n    Follow-up to commit f8ffad69c9f8 (\"bpf: add skb_postpush_rcsum and fix\n    dev_forward_skb occasions\") to fix an issue for dev_queue_xmit() redirect\n    locations which need CHECKSUM_COMPLETE fixups on ingress.\n\n    For the same reasons as described in f8ffad69c9f8 already, we of course\n    also need this here, since dev_queue_xmit() on a veth device will let us\n    end up in the dev_forward_skb() helper again to cross namespaces.\n\n    Latter then calls into skb_postpull_rcsum() to pull out L2 header, so\n    that netif_rx_internal() sees CHECKSUM_COMPLETE as it is expected. That\n    is, CHECKSUM_COMPLETE on ingress covering L2 _payload_, not L2 headers.\n\n    Also here we have to address bpf_redirect() and bpf_clone_redirect().\n\n    Fixes: 3896d655f4d4 (\"bpf: introduce bpf_clone_redirect() helper\")\n    Fixes: 27b29f63058d (\"bpf: add bpf_redirect() helper\")\n    Signed-off-by: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Acked-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 936b09e8df85b8193f0a6e4577323d0e14d19419\nAuthor: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\nDate:   Fri Jun 10 21:19:06 2016 +0200\n\n    bpf: enforce recursion limit on redirects\n\n    Respect the stack\u0027s xmit_recursion limit for calls into dev_queue_xmit().\n    Currently, they are not handeled by the limiter when attached to clsact\u0027s\n    egress parent, for example, and a buggy program redirecting it to the\n    same device again could run into stack overflow eventually. It would be\n    good if we could notify an admin to give him a chance to react. We reuse\n    xmit_recursion instead of having one private to eBPF, so that the stack\u0027s\n    current recursion depth will be taken into account as well. Follow-up to\n    commit 3896d655f4d4 (\"bpf: introduce bpf_clone_redirect() helper\") and\n    27b29f63058d (\"bpf: add bpf_redirect() helper\").\n\n    Signed-off-by: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Acked-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit f401a2efb35493d5ac66c21c4a14fdfd9da56d9e\nAuthor: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\nDate:   Fri Sep 9 02:45:30 2016 +0200\n\n    bpf: add own ctx rewriter on ifindex for clsact progs\n\n    When fetching ifindex, we don\u0027t need to test dev for being NULL since\n    we\u0027re always guaranteed to have a valid dev for clsact programs. Thus,\n    avoid this test in fast path.\n\n    Signed-off-by: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Acked-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 3862495597f4a9de172a87b01c2ebeb4fbb64d05\nAuthor: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\nDate:   Thu Jul 14 18:08:04 2016 +0200\n\n    bpf, perf: split bpf_perf_event_output\n\n    Split the bpf_perf_event_output() helper as a preparation into\n    two parts. The new bpf_perf_event_output() will prepare the raw\n    record itself and test for unknown flags from BPF trace context,\n    where the __bpf_perf_event_output() does the core work. The\n    latter will be reused later on from bpf_event_output() directly.\n\n    Signed-off-by: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Acked-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit ec650fbdacc972f3ec5c570c72a804ccdf2d46bd\nAuthor: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\nDate:   Fri Sep 9 02:45:29 2016 +0200\n\n    bpf: add BPF_SIZEOF and BPF_FIELD_SIZEOF macros\n\n    Add BPF_SIZEOF() and BPF_FIELD_SIZEOF() macros to improve the code a bit\n    which otherwise often result in overly long bytes_to_bpf_size(sizeof())\n    and bytes_to_bpf_size(FIELD_SIZEOF()) lines. So place them into a macro\n    helper instead. Moreover, we currently have a BUILD_BUG_ON(BPF_FIELD_SIZEOF())\n    check in convert_bpf_extensions(), but we should rather make that generic\n    as well and add a BUILD_BUG_ON() test in all BPF_SIZEOF()/BPF_FIELD_SIZEOF()\n    users to detect any rewriter size issues at compile time. Note, there are\n    currently none, but we want to assert that it stays this way.\n\n    Signed-off-by: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Acked-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit f9bf2593b3f0dcd29447d2a9b4ec8f26e8efb8f6\nAuthor: Alexei Starovoitov \u003cast@fb.com\u003e\nDate:   Thu Sep 1 18:37:22 2016 -0700\n\n    bpf: introduce BPF_PROG_TYPE_PERF_EVENT program type\n\n    Introduce BPF_PROG_TYPE_PERF_EVENT programs that can be attached to\n    HW and SW perf events (PERF_TYPE_HARDWARE and PERF_TYPE_SOFTWARE\n    correspondingly in uapi/linux/perf_event.h)\n\n    The program visible context meta structure is\n    struct bpf_perf_event_data {\n        struct pt_regs regs;\n         __u64 sample_period;\n    };\n    which is accessible directly from the program:\n    int bpf_prog(struct bpf_perf_event_data *ctx)\n    {\n      ... ctx-\u003esample_period ...\n      ... ctx-\u003eregs.ip ...\n    }\n\n    The bpf verifier rewrites the accesses into kernel internal\n    struct bpf_perf_event_data_kern which allows changing\n    struct perf_sample_data without affecting bpf programs.\n    New fields can be added to the end of struct bpf_perf_event_data\n    in the future.\n\n    Signed-off-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Acked-by: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 57219c7eeaf3348a87a3bae0647510002dbd976d\nAuthor: Alexei Starovoitov \u003cast@fb.com\u003e\nDate:   Thu Aug 11 18:17:18 2016 -0700\n\n    bpf: allow bpf_get_prandom_u32() to be used in tracing\n\n    bpf_get_prandom_u32() was initially introduced for socket filters\n    and later requested numberous times to be added to tracing bpf programs\n    for the same reason as in socket filters: to be able to randomly\n    select incoming events.\n\n    Signed-off-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Acked-by: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 12892369eecbbe9b3cf6418f907b1d0550e322bc\nAuthor: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\nDate:   Fri Sep 9 02:45:28 2016 +0200\n\n    bpf: minor cleanups in helpers\n\n    Some minor misc cleanups, f.e. use sizeof(__u32) instead of hardcoding\n    and in __bpf_skb_max_len(), I missed that we always have skb-\u003edev valid\n    anyway, so we can drop the unneeded test for dev; also few more other\n    misc bits addressed here.\n\n    Signed-off-by: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Acked-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit a5da0516449f6bf024a50d9afed70b9ffcfd9183\nAuthor: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\nDate:   Thu Aug 18 01:00:41 2016 +0200\n\n    bpf: get rid of cgroup helper related ifdefs\n\n    As recently discussed during the task_under_cgroup_hierarchy() addition,\n    we should get rid of the ifdefs surrounding the bpf_skb_under_cgroup()\n    helper. If related functionality is not built-in, the helper cannot be\n    used anyway, which is also in line with what we do for all other helpers.\n\n    Signed-off-by: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Acked-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 49604d19727ab335184ce50c08cb4787175c0a77\nAuthor: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\nDate:   Thu Aug 18 01:00:40 2016 +0200\n\n    bpf: enable event output helper also for xdp types\n\n    Follow-up to 555c8a8623a3 (\"bpf: avoid stack copy and use skb ctx for\n    event output\") for also adding the event output helper for XDP typed\n    programs. The event output helper has been very useful in particular for\n    debugging or event notification purposes, since it\u0027s much faster and\n    flexible than regular trace printk due to programmatically being able to\n    attach meta data. Same flags structure applies as with tc BPF programs.\n\n    Signed-off-by: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Acked-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 1525487e67bb1f699506af61a734945ec7356245\nAuthor: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\nDate:   Thu Aug 18 01:00:39 2016 +0200\n\n    bpf: add bpf_skb_change_tail helper\n\n    This work adds a bpf_skb_change_tail() helper for tc BPF programs. The\n    basic idea is to expand or shrink the skb in a controlled manner. The\n    eBPF program can then rewrite the rest via helpers like bpf_skb_store_bytes(),\n    bpf_lX_csum_replace() and others rather than passing a raw buffer for\n    writing here.\n\n    bpf_skb_change_tail() is really a slow path helper and intended for\n    replies with f.e. ICMP control messages. Concept is similar to other\n    helpers like bpf_skb_change_proto() helper to keep the helper without\n    protocol specifics and let the BPF program mangle the remaining parts.\n    A flags field has been added and is reserved for now should we extend\n    the helper in future.\n\n    Signed-off-by: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Acked-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 3d0ac151eb1b8dd013bb7a149360959c0ae60f01\nAuthor: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\nDate:   Thu Aug 18 01:00:38 2016 +0200\n\n    bpf: use skb_pkt_type_ok helper in bpf_skb_change_type\n\n    Since we have a skb_pkt_type_ok() helper for checking the type before\n    mangling, make use of it instead of open coding. Follow-up to commit\n    8b10cab64c13 (\"net: simplify and make pkt_type_ok() available for other\n    users\") that came in after d2485c4242a8 (\"bpf: add bpf_skb_change_type\n    helper\").\n\n    Signed-off-by: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Acked-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit ab67f4e1fecde3adbc512baa7360cedbc44deac0\nAuthor: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\nDate:   Thu Aug 11 21:38:37 2016 +0200\n\n    bpf: fix write helpers with regards to non-linear parts\n\n    Fix the bpf_try_make_writable() helper and all call sites we have in BPF,\n    it\u0027s currently defect with regards to skbs when the write_len spans into\n    non-linear parts, no matter if cloned or not.\n\n    There are multiple issues at once. First, using skb_store_bits() is not\n    correct since even if we have a cloned skb, page frags can still be shared.\n    To really make them private, we need to pull them in via __pskb_pull_tail()\n    first, which also gets us a private head via pskb_expand_head() implicitly.\n\n    This is for helpers like bpf_skb_store_bytes(), bpf_l3_csum_replace(),\n    bpf_l4_csum_replace(). Really, the only thing reasonable and working here\n    is to call skb_ensure_writable() before any write operation. Meaning, via\n    pskb_may_pull() it makes sure that parts we want to access are pulled in and\n    if not does so plus unclones the skb implicitly. If our write_len still fits\n    the headlen and we\u0027re cloned and our header of the clone is not writable,\n    then we need to make a private copy via pskb_expand_head(). skb_store_bits()\n    is a bit misleading and only safe to store into non-linear data in different\n    contexts such as 357b40a18b04 (\"[IPV6]: IPV6_CHECKSUM socket option can\n    corrupt kernel memory\").\n\n    For above BPF helper functions, it means after fixed bpf_try_make_writable(),\n    we\u0027ve pulled in enough, so that we operate always based on skb-\u003edata. Thus,\n    the call to skb_header_pointer() and skb_store_bits() becomes superfluous.\n    In bpf_skb_store_bytes(), the len check is unnecessary too since it can\n    only pass in maximum of BPF stack size, so adding offset is guaranteed to\n    never overflow. Also bpf_l3/4_csum_replace() helpers must test for proper\n    offset alignment since they use __sum16 pointer for writing resulting csum.\n\n    The remaining helpers that change skb data not discussed here yet are\n    bpf_skb_vlan_push(), bpf_skb_vlan_pop() and bpf_skb_change_proto(). The\n    vlan helpers internally call either skb_ensure_writable() (pop case) and\n    skb_cow_head() (push case, for head expansion), respectively. Similarly,\n    bpf_skb_proto_xlat() takes care to not mangle page frags.\n\n    Fixes: 608cd71a9c7c (\"tc: bpf: generalize pedit action\")\n    Fixes: 91bc4822c3d6 (\"tc: bpf: add checksum helpers\")\n    Fixes: 3697649ff29e (\"bpf: try harder on clones when writing into skb\")\n    Signed-off-by: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Acked-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 1f476a566de499a6e11f27f46a116ee0fcf60b9d\nAuthor: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\nDate:   Tue Sep 20 00:26:14 2016 +0200\n\n    bpf: add test cases for direct packet access\n\n    Add couple of test cases for direct write and the negative size issue, and\n    also adjust the direct packet access test4 since it asserts that writes are\n    not possible, but since we\u0027ve just added support for writes, we need to\n    invert the verdict to ACCEPT, of course. Summary: 133 PASSED, 0 FAILED.\n\n    Signed-off-by: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Acked-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 8e932e250e06f391e708b92137c5a23d22938619\nAuthor: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\nDate:   Thu Sep 8 01:03:42 2016 +0200\n\n    bpf: fix range propagation on direct packet access\n\n    LLVM can generate code that tests for direct packet access via\n    skb-\u003edata/data_end in a way that currently gets rejected by the\n    verifier, example:\n\n      [...]\n       7: (61) r3 \u003d *(u32 *)(r6 +80)\n       8: (61) r9 \u003d *(u32 *)(r6 +76)\n       9: (bf) r2 \u003d r9\n      10: (07) r2 +\u003d 54\n      11: (3d) if r3 \u003e\u003d r2 goto pc+12\n       R1\u003dinv R2\u003dpkt(id\u003d0,off\u003d54,r\u003d0) R3\u003dpkt_end R4\u003dinv R6\u003dctx\n       R9\u003dpkt(id\u003d0,off\u003d0,r\u003d0) R10\u003dfp\n      12: (18) r4 \u003d 0xffffff7a\n      14: (05) goto pc+430\n      [...]\n\n      from 11 to 24: R1\u003dinv R2\u003dpkt(id\u003d0,off\u003d54,r\u003d0) R3\u003dpkt_end R4\u003dinv\n                     R6\u003dctx R9\u003dpkt(id\u003d0,off\u003d0,r\u003d0) R10\u003dfp\n      24: (7b) *(u64 *)(r10 -40) \u003d r1\n      25: (b7) r1 \u003d 0\n      26: (63) *(u32 *)(r6 +56) \u003d r1\n      27: (b7) r2 \u003d 40\n      28: (71) r8 \u003d *(u8 *)(r9 +20)\n      invalid access to packet, off\u003d20 size\u003d1, R9(id\u003d0,off\u003d0,r\u003d0)\n\n    The reason why this gets rejected despite a proper test is that we\n    currently call find_good_pkt_pointers() only in case where we detect\n    tests like rX \u003e pkt_end, where rX is of type pkt(id\u003dY,off\u003dZ,r\u003d0) and\n    derived, for example, from a register of type pkt(id\u003dY,off\u003d0,r\u003d0)\n    pointing to skb-\u003edata. find_good_pkt_pointers() then fills the range\n    in the current branch to pkt(id\u003dY,off\u003d0,r\u003dZ) on success.\n\n    For above case, we need to extend that to recognize pkt_end \u003e\u003d rX\n    pattern and mark the other branch that is taken on success with the\n    appropriate pkt(id\u003dY,off\u003d0,r\u003dZ) type via find_good_pkt_pointers().\n    Since eBPF operates on BPF_JGT (\u003e) and BPF_JGE (\u003e\u003d), these are the\n    only two practical options to test for from what LLVM could have\n    generated, since there\u0027s no such thing as BPF_JLT (\u003c) or BPF_JLE (\u003c\u003d)\n    that we would need to take into account as well.\n\n    After the fix:\n\n      [...]\n       7: (61) r3 \u003d *(u32 *)(r6 +80)\n       8: (61) r9 \u003d *(u32 *)(r6 +76)\n       9: (bf) r2 \u003d r9\n      10: (07) r2 +\u003d 54\n      11: (3d) if r3 \u003e\u003d r2 goto pc+12\n       R1\u003dinv R2\u003dpkt(id\u003d0,off\u003d54,r\u003d0) R3\u003dpkt_end R4\u003dinv R6\u003dctx\n       R9\u003dpkt(id\u003d0,off\u003d0,r\u003d0) R10\u003dfp\n      12: (18) r4 \u003d 0xffffff7a\n      14: (05) goto pc+430\n      [...]\n\n      from 11 to 24: R1\u003dinv R2\u003dpkt(id\u003d0,off\u003d54,r\u003d54) R3\u003dpkt_end R4\u003dinv\n                     R6\u003dctx R9\u003dpkt(id\u003d0,off\u003d0,r\u003d54) R10\u003dfp\n      24: (7b) *(u64 *)(r10 -40) \u003d r1\n      25: (b7) r1 \u003d 0\n      26: (63) *(u32 *)(r6 +56) \u003d r1\n      27: (b7) r2 \u003d 40\n      28: (71) r8 \u003d *(u8 *)(r9 +20)\n      29: (bf) r1 \u003d r8\n      30: (25) if r8 \u003e 0x3c goto pc+47\n       R1\u003dinv56 R2\u003dimm40 R3\u003dpkt_end R4\u003dinv R6\u003dctx R8\u003dinv56\n       R9\u003dpkt(id\u003d0,off\u003d0,r\u003d54) R10\u003dfp\n      31: (b7) r1 \u003d 1\n      [...]\n\n    Verifier test cases are also added in this work, one that demonstrates\n    the mentioned example here and one that tries a bad packet access for\n    the current/fall-through branch (the one with types pkt(id\u003dX,off\u003dY,r\u003d0),\n    pkt(id\u003dX,off\u003d0,r\u003d0)), then a case with good and bad accesses, and two\n    with both test variants (\u003e, \u003e\u003d).\n\n    Fixes: 969bf05eb3ce (\"bpf: direct packet access\")\n    Signed-off-by: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Acked-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 873b1be73c1d8a7d86a99c7005f6e7cec8568690\nAuthor: Aaron Yue \u003chaoxuany@fb.com\u003e\nDate:   Thu Aug 11 18:17:17 2016 -0700\n\n    samples/bpf: add verifier tests for the helper access to the packet\n\n    test various corner cases of the helper function access to the packet\n    via crafted XDP programs.\n\n    Signed-off-by: Aaron Yue \u003chaoxuany@fb.com\u003e\n    Signed-off-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Acked-by: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit dec7ab59647a8d197cc252b027f5533d2befb9a3\nAuthor: Alexei Starovoitov \u003cast@fb.com\u003e\nDate:   Thu May 5 19:49:15 2016 -0700\n\n    samples/bpf: add verifier tests\n\n    add few tests for \"pointer to packet\" logic of the verifier\n\n    Signed-off-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 5d13af94b66a9fdd4c98098c13baa79bba8de1ed\nAuthor: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\nDate:   Wed Apr 13 00:10:54 2016 +0200\n\n    bpf, samples: add test cases for raw stack\n\n    This adds test cases mostly around ARG_PTR_TO_RAW_STACK to check the\n    verifier behaviour.\n\n      [...]\n      #84 raw_stack: no skb_load_bytes OK\n      #85 raw_stack: skb_load_bytes, no init OK\n      #86 raw_stack: skb_load_bytes, init OK\n      #87 raw_stack: skb_load_bytes, spilled regs around bounds OK\n      #88 raw_stack: skb_load_bytes, spilled regs corruption OK\n      #89 raw_stack: skb_load_bytes, spilled regs corruption 2 OK\n      #90 raw_stack: skb_load_bytes, spilled regs + data OK\n      #91 raw_stack: skb_load_bytes, invalid access 1 OK\n      #92 raw_stack: skb_load_bytes, invalid access 2 OK\n      #93 raw_stack: skb_load_bytes, invalid access 3 OK\n      #94 raw_stack: skb_load_bytes, invalid access 4 OK\n      #95 raw_stack: skb_load_bytes, invalid access 5 OK\n      #96 raw_stack: skb_load_bytes, invalid access 6 OK\n      #97 raw_stack: skb_load_bytes, large access OK\n      Summary: 98 PASSED, 0 FAILED\n\n    Signed-off-by: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Acked-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 7e818ab049fd6f3f945baadb7ec9c9e71d0f6760\nAuthor: Alexei Starovoitov \u003cast@fb.com\u003e\nDate:   Mon Mar 7 21:57:20 2016 -0800\n\n    samples/bpf: add map_flags to bpf loader\n\n    note old loader is compatible with new kernel.\n    map_flags are optional\n\n    Signed-off-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit bb5568f8e3f8be87500558fc7409b5a2180e8196\nAuthor: tom.leiming@gmail.com \u003ctom.leiming@gmail.com\u003e\nDate:   Mon Feb 1 22:39:57 2016 -0800\n\n    samples/bpf: unit test for BPF_MAP_TYPE_PERCPU_ARRAY\n\n    A sanity test for BPF_MAP_TYPE_PERCPU_ARRAY\n\n    Signed-off-by: Ming Lei \u003ctom.leiming@gmail.com\u003e\n    Signed-off-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 8f013495498b22664c2e215696900b9a4787bd0b\nAuthor: Alexei Starovoitov \u003cast@fb.com\u003e\nDate:   Mon Mar 7 21:57:18 2016 -0800\n\n    samples/bpf: make map creation more verbose\n\n    map creation is typically the first one to fail when rlimits are\n    too low, not enough memory, etc\n    Make this failure scenario more verbose\n\n    Signed-off-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 71e38cc3478ffe6f15733a649c70db307e75fad6\nAuthor: Martin KaFai Lau \u003ckafai@fb.com\u003e\nDate:   Mon Feb 1 22:39:56 2016 -0800\n\n    samples/bpf: unit test for BPF_MAP_TYPE_PERCPU_HASH\n\n    A sanity test for BPF_MAP_TYPE_PERCPU_HASH.\n\n    Signed-off-by: Martin KaFai Lau \u003ckafai@fb.com\u003e\n    Signed-off-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit dd858c4846883fbde61bfa221f85ee04549254b5\nAuthor: Alexei Starovoitov \u003cast@fb.com\u003e\nDate:   Thu Sep 1 18:37:23 2016 -0700\n\n    bpf: perf_event progs should only use preallocated maps\n\n    Make sure that BPF_PROG_TYPE_PERF_EVENT programs only use\n    preallocated hash maps, since doing memory allocation\n    in overflow_handler can crash depending on where nmi got triggered.\n\n    Signed-off-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Acked-by: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 1b9b54d940e2aa7c51b97e8e5f26a1c332f83f96\nAuthor: Alexei Starovoitov \u003cast@fb.com\u003e\nDate:   Thu Sep 1 18:37:21 2016 -0700\n\n    bpf: support 8-byte metafield access\n\n    The verifier supported only 4-byte metafields in\n    struct __sk_buff and struct xdp_md. The metafields in upcoming\n    struct bpf_perf_event are 8-byte to match register width in struct pt_regs.\n    Teach verifier to recognize 8-byte metafield access.\n    The patch doesn\u0027t affect safety of sockets and xdp programs.\n    They check for 4-byte only ctx access before these conditions are hit.\n\n    Signed-off-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Acked-by: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit e317791173a2653deb2117cf6c6ec2525803c78b\nAuthor: Alexei Starovoitov \u003cast@fb.com\u003e\nDate:   Thu Aug 11 18:17:16 2016 -0700\n\n    bpf: allow helpers access the packet directly\n\n    The helper functions like bpf_map_lookup_elem(map, key) were only\n    allowing \u0027key\u0027 to point to the initialized stack area.\n    That is causing performance degradation when programs need to process\n    millions of packets per second and need to copy contents of the packet\n    into the stack just to pass the stack pointer into the lookup() function.\n    Allow such helpers read from the packet directly.\n    All helpers that expect ARG_PTR_TO_MAP_KEY, ARG_PTR_TO_MAP_VALUE,\n    ARG_PTR_TO_STACK assume byte aligned pointer, so no alignment concerns,\n    only need to check that helper will not be accessing beyond\n    the packet range verified by the prior \u0027if (ptr \u003c data_end)\u0027 condition.\n    For now allow this feature for XDP programs only. Later it can be\n    relaxed for the clsact programs as well.\n\n    Signed-off-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Acked-by: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit ee956448cd428ad0d5bb950753b4030dff4fa385\nAuthor: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\nDate:   Fri Aug 12 22:17:17 2016 +0200\n\n    bpf: fix bpf_skb_in_cgroup helper naming\n\n    While hashing out BPF\u0027s current_task_under_cgroup helper bits, it came\n    to discussion that the skb_in_cgroup helper name was suboptimally chosen.\n\n    Tejun says:\n\n      So, I think in_cgroup should mean that the object is in that\n      particular cgroup while under_cgroup in the subhierarchy of that\n      cgroup. Let\u0027s rename the other subhierarchy test to under too. I\n      think that\u0027d be a lot less confusing going forward.\n\n      [...]\n\n      It\u0027s more intuitive and gives us the room to implement the real\n      \"in\" test if ever necessary in the future.\n\n    Since this touches uapi bits, we need to change this as long as v4.8\n    is not yet officially released. Thus, change the helper enum and rename\n    related bits.\n\n    Fixes: 4a482f34afcc (\"cgroup: bpf: Add bpf_skb_in_cgroup_proto\")\n    Reference: http://patchwork.ozlabs.org/patch/658500/\n    Suggested-by: Sargun Dhillon \u003csargun@sargun.me\u003e\n    Suggested-by: Tejun Heo \u003ctj@kernel.org\u003e\n    Signed-off-by: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Acked-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 9a6848be325ec1f65ecda69f29a0f7de6ef11583\nAuthor: Martin KaFai Lau \u003ckafai@fb.com\u003e\nDate:   Thu Jun 30 10:28:45 2016 -0700\n\n    cgroup: bpf: Add an example to do cgroup checking in BPF\n\n    test_cgrp2_array_pin.c:\n    A userland program that creates a bpf_map (BPF_MAP_TYPE_GROUP_ARRAY),\n    pouplates/updates it with a cgroup2\u0027s backed fd and pins it to a\n    bpf-fs\u0027s file.  The pinned file can be loaded by tc and then used\n    by the bpf prog later.  This program can also update an existing pinned\n    array and it could be useful for debugging/testing purpose.\n\n    test_cgrp2_tc_kern.c:\n    A bpf prog which should be loaded by tc.  It is to demonstrate\n    the usage of bpf_skb_in_cgroup.\n\n    test_cgrp2_tc.sh:\n    A script that glues the test_cgrp2_array_pin.c and\n    test_cgrp2_tc_kern.c together.  The idea is like:\n    1. Load the test_cgrp2_tc_kern.o by tc\n    2. Use test_cgrp2_array_pin.c to populate a BPF_MAP_TYPE_CGROUP_ARRAY\n       with a cgroup fd\n    3. Do a \u0027ping -6 ff02::1%ve\u0027 to ensure the packet has been\n       dropped because of a match on the cgroup\n\n    Most of the lines in test_cgrp2_tc.sh is the boilerplate\n    to setup the cgroup/bpf-fs/net-devices/netns...etc.  It is\n    not bulletproof on errors but should work well enough and\n    give enough debug info if things did not go well.\n\n    Signed-off-by: Martin KaFai Lau \u003ckafai@fb.com\u003e\n    Cc: Alexei Starovoitov \u003cast@fb.com\u003e\n    Cc: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Cc: Tejun Heo \u003ctj@kernel.org\u003e\n    Acked-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 554c0c2a1e638f804abd90ec1a6465b944ebc43a\nAuthor: Alexei Starovoitov \u003cast@fb.com\u003e\nDate:   Thu May 5 19:49:14 2016 -0700\n\n    samples/bpf: add \u0027pointer to packet\u0027 tests\n\n    parse_simple.c - packet parser exapmle with single length check that\n    filters out udp packets for port 9\n\n    parse_varlen.c - variable length parser that understand multiple vlan headers,\n    ipip, ipip6 and ip options to filter out udp or tcp packets on port 9.\n    The packet is parsed layer by layer with multitple length checks.\n\n    parse_ldabs.c - classic style of packet parsing using LD_ABS instruction.\n    Same functionality as parse_simple.\n\n    simple \u003d 24.1Mpps per core\n    varlen \u003d 22.7Mpps\n    ldabs  \u003d 21.4Mpps\n\n    Parser with LD_ABS instructions is slower than full direct access parser\n    which does more packet accesses and checks.\n\n    These examples demonstrate the choice bpf program authors can make between\n    flexibility of the parser vs speed.\n\n    Signed-off-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 05c11618658607b7b87630b177ae360cd3832108\nAuthor: Alexei Starovoitov \u003cast@fb.com\u003e\nDate:   Wed Apr 6 18:43:31 2016 -0700\n\n    samples/bpf: add tracepoint vs kprobe performance tests\n\n    the first microbenchmark does\n    fd\u003dopen(\"/proc/self/comm\");\n    for() {\n      write(fd, \"test\");\n    }\n    and on 4 cpus in parallel:\n                                          writes per sec\n    base (no tracepoints, no kprobes)         930k\n    with kprobe at __set_task_comm()          420k\n    with tracepoint at task:task_rename       730k\n\n    For kprobe + full bpf program manully fetches oldcomm, newcomm via bpf_probe_read.\n    For tracepint bpf program does nothing, since arguments are copied by tracepoint.\n\n    2nd microbenchmark does:\n    fd\u003dopen(\"/dev/urandom\");\n    for() {\n      read(fd, buf);\n    }\n    and on 4 cpus in parallel:\n                                           reads per sec\n    base (no tracepoints, no kprobes)         300k\n    with kprobe at urandom_read()             279k\n    with tracepoint at random:urandom_read    290k\n\n    bpf progs attached to kprobe and tracepoint are noop.\n\n    Signed-off-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 11304e9b3f6f30bde8203e8a40bc07d1484026a1\nAuthor: Alexei Starovoitov \u003cast@fb.com\u003e\nDate:   Tue Mar 8 15:07:54 2016 -0800\n\n    samples/bpf: add map performance test\n\n    performance tests for hash map and per-cpu hash map\n    with and without pre-allocation\n\n    Signed-off-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 619046c6d05171bcb9289345c661a127e5838eae\nAuthor: Alexei Starovoitov \u003cast@fb.com\u003e\nDate:   Tue Mar 8 15:07:52 2016 -0800\n\n    samples/bpf: add bpf map stress test\n\n    this test calls bpf programs from different contexts:\n    from inside of slub, from rcu, from pretty much everywhere,\n    since it kprobes all spin_lock functions.\n    It stresses the bpf hash and percpu map pre-allocation,\n    deallocation logic and call_rcu mechanisms.\n    User space part adding more stress by walking and deleting map elements.\n\n    Note that due to nature bpf_load.c the earlier kprobe+bpf programs are\n    already active while loader loads new programs, creates new kprobes and\n    attaches them.\n\n    Signed-off-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit edb511a8eed3a35d263e3b41b656d0d2e042b2f1\nAuthor: Sargun Dhillon \u003csargun@sargun.me\u003e\nDate:   Fri Aug 12 08:56:52 2016 -0700\n\n    bpf: Add bpf_current_task_under_cgroup helper\n\n    This adds a bpf helper that\u0027s similar to the skb_in_cgroup helper to check\n    whether the probe is currently executing in the context of a specific\n    subset of the cgroupsv2 hierarchy. It does this based on membership test\n    for a cgroup arraymap. It is invalid to call this in an interrupt, and\n    it\u0027ll return an error. The helper is primarily to be used in debugging\n    activities for containers, where you may have multiple programs running in\n    a given top-level \"container\".\n\n    Signed-off-by: Sargun Dhillon \u003csargun@sargun.me\u003e\n    Cc: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Cc: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Cc: Tejun Heo \u003ctj@kernel.org\u003e\n    Acked-by: Tejun Heo \u003ctj@kernel.org\u003e\n    Acked-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Acked-by: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 6d795673a81321dddaabd489fc5cb797be9bb179\nAuthor: Sargun Dhillon \u003csargun@sargun.me\u003e\nDate:   Mon Jul 25 05:54:46 2016 -0700\n\n    bpf: Add bpf_probe_write_user BPF helper to be called in tracers\n\n    This allows user memory to be written to during the course of a kprobe.\n    It shouldn\u0027t be used to implement any kind of security mechanism\n    because of TOC-TOU attacks, but rather to debug, divert, and\n    manipulate execution of semi-cooperative processes.\n\n    Although it uses probe_kernel_write, we limit the address space\n    the probe can write into by checking the space with access_ok.\n    We do this as opposed to calling copy_to_user directly, in order\n    to avoid sleeping. In addition we ensure the threads\u0027s current fs\n    / segment is USER_DS and the thread isn\u0027t exiting nor a kernel thread.\n\n    Given this feature is meant for experiments, and it has a risk of\n    crashing the system, and running programs, we print a warning on\n    when a proglet that attempts to use this helper is installed,\n    along with the pid and process name.\n\n    Signed-off-by: Sargun Dhillon \u003csargun@sargun.me\u003e\n    Cc: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Cc: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Acked-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit c55d6a90647e422426377f262f39fb6d5fe756b2\nAuthor: Alexei Starovoitov \u003cast@fb.com\u003e\nDate:   Wed Feb 17 19:58:59 2016 -0800\n\n    samples/bpf: offwaketime example\n\n    This is simplified version of Brendan Gregg\u0027s offwaketime:\n    This program shows kernel stack traces and task names that were blocked and\n    \"off-CPU\", along with the stack traces and task names for the threads that woke\n    them, and the total elapsed time from when they blocked to when they were woken\n    up. The combined stacks, task names, and total time is summarized in kernel\n    context for efficiency.\n\n    Example:\n    $ sudo ./offwaketime | flamegraph.pl \u003e demo.svg\n    Open demo.svg in the browser as FlameGraph visualization.\n\n    Signed-off-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit b7d3d530ad8d254eddd35367b1deede6c67dc2fc\nAuthor: Andrew Morton \u003cakpm@linux-foundation.org\u003e\nDate:   Mon Jul 18 15:50:58 2016 -0700\n\n    kernel/trace/bpf_trace.c: work around gcc-4.4.4 anon union initialization bug\n\n    kernel/trace/bpf_trace.c: In function \u0027bpf_event_output\u0027:\n    kernel/trace/bpf_trace.c:312: error: unknown field \u0027next\u0027 specified in initializer\n    kernel/trace/bpf_trace.c:312: warning: missing braces around initializer\n    kernel/trace/bpf_trace.c:312: warning: (near initialization for \u0027raw.frag.\u003canonymous\u003e\u0027)\n\n    Fixes: 555c8a8623a3a87 (\"bpf: avoid stack copy and use skb ctx for event output\")\n    Acked-by: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Cc: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Cc: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Andrew Morton \u003cakpm@linux-foundation.org\u003e\n    Acked-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit c26cfd27cfcd02b861e76cd3b161ffca56a50c4f\nAuthor: Alexei Starovoitov \u003cast@fb.com\u003e\nDate:   Wed Jul 6 22:38:36 2016 -0700\n\n    bpf: introduce bpf_get_current_task() helper\n\n    over time there were multiple requests to access different data\n    structures and fields of task_struct current, so finally add\n    the helper to access \u0027current\u0027 as-is. Tracing bpf programs will do\n    the rest of walking the pointers via bpf_probe_read().\n    Note that current can be null and bpf program has to deal it with,\n    but even dumb passing null into bpf_probe_read() is still safe.\n\n    Suggested-by: Brendan Gregg \u003cbrendan.d.gregg@gmail.com\u003e\n    Signed-off-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Acked-by: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit fbea0d499e0041c6ecffc1e0561b7f7648f63873\nAuthor: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\nDate:   Sun Jul 3 01:28:47 2016 +0200\n\n    bpf: add bpf_get_hash_recalc helper\n\n    If skb_clear_hash() was invoked due to mangling of relevant headers and\n    BPF program needs skb-\u003ehash later on, we can add a helper to trigger hash\n    recalculation via bpf_get_hash_recalc().\n\n    The helper will return the newly retrieved hash directly, but later access\n    can also be done via skb context again through skb-\u003ehash directly (inline)\n    without needing to call the helper once more.\n\n    Signed-off-by: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Acked-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit d46c43b45b9139adbffdeacb03fd14eca7cc5220\nAuthor: Alexei Starovoitov \u003cast@fb.com\u003e\nDate:   Fri Aug 5 14:01:27 2016 -0700\n\n    bpf: restore behavior of bpf_map_update_elem\n\n    The introduction of pre-allocated hash elements inadvertently broke\n    the behavior of bpf hash maps where users expected to call\n    bpf_map_update_elem() without considering that the map can be full.\n    Some programs do:\n    old_value \u003d bpf_map_lookup_elem(map, key);\n    if (old_value) {\n      ... prepare new_value on stack ...\n      bpf_map_update_elem(map, key, new_value);\n    }\n    Before pre-alloc the update() for existing element would work even\n    in \u0027map full\u0027 condition. Restore this behavior.\n\n    The above program could have updated old_value in place instead of\n    update() which would be faster and most programs use that approach,\n    but sometimes the values are large and the programs use update()\n    helper to do atomic replacement of the element.\n    Note we cannot simply update element\u0027s value in-place like percpu\n    hash map does and have to allocate extra num_possible_cpu elements\n    and use this extra reserve when the map is full.\n\n    Fixes: 6c9059817432 (\"bpf: pre-allocate hash map elements\")\n    Signed-off-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 4527c21762060ff534949c16348edb1cb0c53ce3\nAuthor: Jakub Kicinski \u003cjakub.kicinski@netronome.com\u003e\nDate:   Tue Aug 2 16:12:14 2016 +0100\n\n    bpf: fix method of PTR_TO_PACKET reg id generation\n\n    Using per-register incrementing ID can lead to\n    find_good_pkt_pointers() confusing registers which\n    have completely different values.  Consider example:\n\n    0: (bf) r6 \u003d r1\n    1: (61) r8 \u003d *(u32 *)(r6 +76)\n    2: (61) r0 \u003d *(u32 *)(r6 +80)\n    3: (bf) r7 \u003d r8\n    4: (07) r8 +\u003d 32\n    5: (2d) if r8 \u003e r0 goto pc+9\n     R0\u003dpkt_end R1\u003dctx R6\u003dctx R7\u003dpkt(id\u003d0,off\u003d0,r\u003d32) R8\u003dpkt(id\u003d0,off\u003d32,r\u003d32) R10\u003dfp\n    6: (bf) r8 \u003d r7\n    7: (bf) r9 \u003d r7\n    8: (71) r1 \u003d *(u8 *)(r7 +0)\n    9: (0f) r8 +\u003d r1\n    10: (71) r1 \u003d *(u8 *)(r7 +1)\n    11: (0f) r9 +\u003d r1\n    12: (07) r8 +\u003d 32\n    13: (2d) if r8 \u003e r0 goto pc+1\n     R0\u003dpkt_end R1\u003dinv56 R6\u003dctx R7\u003dpkt(id\u003d0,off\u003d0,r\u003d32) R8\u003dpkt(id\u003d1,off\u003d32,r\u003d32) R9\u003dpkt(id\u003d1,off\u003d0,r\u003d32) R10\u003dfp\n    14: (71) r1 \u003d *(u8 *)(r9 +16)\n    15: (b7) r7 \u003d 0\n    16: (bf) r0 \u003d r7\n    17: (95) exit\n\n    We need to get a UNKNOWN_VALUE with imm to force id\n    generation so lines 0-5 make r7 a valid packet pointer.\n    We then read two different bytes from the packet and\n    add them to copies of the constructed packet pointer.\n    r8 (line 9) and r9 (line 11) will get the same id of 1,\n    independently.  When either of them is validated (line\n    13) - find_good_pkt_pointers() will also mark the other\n    as safe.  This leads to access on line 14 being mistakenly\n    considered safe.\n\n    Fixes: 969bf05eb3ce (\"bpf: direct packet access\")\n    Signed-off-by: Jakub Kicinski \u003cjakub.kicinski@netronome.com\u003e\n    Acked-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Acked-by: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit bb13276696f4411d2401df5cf25f26bbe82d96ee\nAuthor: Brenden Blanco \u003cbblanco@plumgrid.com\u003e\nDate:   Tue Jul 19 12:16:56 2016 -0700\n\n    bpf: enable direct packet data write for xdp progs\n\n    For forwarding to be effective, XDP programs should be allowed to\n    rewrite packet data.\n\n    This requires that the drivers supporting XDP must all map the packet\n    memory as TODEVICE or BIDIRECTIONAL before invoking the program.\n\n    Signed-off-by: Brenden Blanco \u003cbblanco@plumgrid.com\u003e\n    Acked-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 59611c698e67d38393f4c71322a5465d3c4a6920\nAuthor: Brenden Blanco \u003cbblanco@plumgrid.com\u003e\nDate:   Tue Jul 19 12:16:47 2016 -0700\n\n    bpf: add XDP prog type for early driver filter\n\n    Add a new bpf prog type that is intended to run in early stages of the\n    packet rx path. Only minimal packet metadata will be available, hence a\n    new context type, struct xdp_md, is exposed to userspace. So far only\n    expose the packet start and end pointers, and only in read mode.\n\n    An XDP program must return one of the well known enum values, all other\n    return codes are reserved for future use. Unfortunately, this\n    restriction is hard to enforce at verification time, so take the\n    approach of warning at runtime when such programs are encountered. Out\n    of bounds return codes should alias to XDP_ABORTED.\n\n    Signed-off-by: Brenden Blanco \u003cbblanco@plumgrid.com\u003e\n    Acked-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 291e76e5d2af73bfc0d3ce57e72d0e6ad691266c\nAuthor: Alexei Starovoitov \u003cast@fb.com\u003e\nDate:   Thu May 5 19:49:12 2016 -0700\n\n    bpf: wire in data and data_end for cls_act_bpf\n\n    allow cls_bpf and act_bpf programs access skb-\u003edata and skb-\u003edata_end pointers.\n    The bpf helpers that change skb-\u003edata need to update data_end pointer as well.\n    The verifier checks that programs always reload data, data_end pointers\n    after calls to such bpf helpers.\n    We cannot add \u0027data_end\u0027 pointer to struct qdisc_skb_cb directly,\n    since it\u0027s embedded as-is by infiniband ipoib, so wrapper struct is needed.\n\n    Signed-off-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Acked-by: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit f28b5534232c21bbac6b8abc922c5c6fd355a892\nAuthor: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\nDate:   Wed Jan 6 22:32:16 2016 +0100\n\n    bpf: cleanup bpf_prog_run_{save,clear}_cb helpers\n\n    Move the details behind the cb[] access into a small helper to decouple\n    and make them generic for bpf_prog_run_save_cb()/bpf_prog_run_clear_cb()\n    that was introduced via commit ff936a04e5f2 (\"bpf: fix cb access in socket\n    filter programs\"). Also add a comment to better clarify what is done in\n    bpf_skb_cb().\n\n    Signed-off-by: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Acked-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit bb9f86321da94643118a9183610874070b66659b\nAuthor: Brenden Blanco \u003cbblanco@plumgrid.com\u003e\nDate:   Tue Jul 19 12:16:46 2016 -0700\n\n    bpf: add bpf_prog_add api for bulk prog refcnt\n\n    A subsystem may need to store many copies of a bpf program, each\n    deserving its own reference. Rather than requiring the caller to loop\n    one by one (with possible mid-loop failure), add a bulk bpf_prog_add\n    api.\n\n    Signed-off-by: Brenden Blanco \u003cbblanco@plumgrid.com\u003e\n    Acked-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit bd0470638e184265fa12286af3d2794b8adfa033\nAuthor: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\nDate:   Sat Jul 16 01:15:55 2016 +0200\n\n    bpf: bpf_event_entry_gen\u0027s alloc needs to be in atomic context\n\n    Should have been obvious, only called from bpf() syscall via map_update_elem()\n    that calls bpf_fd_array_map_update_elem() under RCU read lock and thus this\n    must also be in GFP_ATOMIC, of course.\n\n    Fixes: 3b1efb196eee (\"bpf, maps: flush own entries on perf map release\")\n    Signed-off-by: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Acked-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit dd0994cece4f1885054b6394a890325c41ab6601\nAuthor: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\nDate:   Thu Jul 14 18:08:05 2016 +0200\n\n    bpf: avoid stack copy and use skb ctx for event output\n\n    This work addresses a couple of issues bpf_skb_event_output()\n    helper currently has: i) We need two copies instead of just a\n    single one for the skb data when it should be part of a sample.\n    The data can be non-linear and thus needs to be extracted via\n    bpf_skb_load_bytes() helper first, and then copied once again\n    into the ring buffer slot. ii) Since bpf_skb_load_bytes()\n    currently needs to be used first, the helper needs to see a\n    constant size on the passed stack buffer to make sure BPF\n    verifier can do sanity checks on it during verification time.\n    Thus, just passing skb-\u003elen (or any other non-constant value)\n    wouldn\u0027t work, but changing bpf_skb_load_bytes() is also not\n    the proper solution, since the two copies are generally still\n    needed. iii) bpf_skb_load_bytes() is just for rather small\n    buffers like headers, since they need to sit on the limited\n    BPF stack anyway. Instead of working around in bpf_skb_load_bytes(),\n    this work improves the bpf_skb_event_output() helper to address\n    all 3 at once.\n\n    We can make use of the passed in skb context that we have in\n    the helper anyway, and use some of the reserved flag bits as\n    a length argument. The helper will use the new __output_custom()\n    facility from perf side with bpf_skb_copy() as callback helper\n    to walk and extract the data. It will pass the data for setup\n    to bpf_event_output(), which generates and pushes the raw record\n    with an additional frag part. The linear data used in the first\n    frag of the record serves as programmatically defined meta data\n    passed along with the appended sample.\n\n    Signed-off-by: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Acked-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 1a4b7f9572753320d7f689c3b5452238b38444f0\nAuthor: Paul Gortmaker \u003cpaul.gortmaker@windriver.com\u003e\nDate:   Mon Jul 11 12:51:01 2016 -0400\n\n    bpf: make inode code explicitly non-modular\n\n    The Kconfig currently controlling compilation of this code is:\n\n    init/Kconfig:config BPF_SYSCALL\n    init/Kconfig:   bool \"Enable bpf() system call\"\n\n    ...meaning that it currently is not being built as a module by anyone.\n\n    Lets remove the couple traces of modular infrastructure use, so that\n    when reading the driver there is no doubt it is builtin-only.\n\n    Note that MODULE_ALIAS is a no-op for non-modular code.\n\n    We replace module.h with init.h since the file does use __init.\n\n    Cc: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Cc: netdev@vger.kernel.org\n    Signed-off-by: Paul Gortmaker \u003cpaul.gortmaker@windriver.com\u003e\n    Acked-by: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 50029f37fa71f61c3e3477ea6861414981d212eb\nAuthor: Martin KaFai Lau \u003ckafai@fb.com\u003e\nDate:   Thu Jun 30 10:28:44 2016 -0700\n\n    cgroup: bpf: Add bpf_skb_in_cgroup_proto\n\n    Adds a bpf helper, bpf_skb_in_cgroup, to decide if a skb-\u003esk\n    belongs to a descendant of a cgroup2.  It is similar to the\n    feature added in netfilter:\n    commit c38c4597e4bf (\"netfilter: implement xt_cgroup cgroup2 path match\")\n\n    The user is expected to populate a BPF_MAP_TYPE_CGROUP_ARRAY\n    which will be used by the bpf_skb_in_cgroup.\n\n    Modifications to the bpf verifier is to ensure BPF_MAP_TYPE_CGROUP_ARRAY\n    and bpf_skb_in_cgroup() are always used together.\n\n    Signed-off-by: Martin KaFai Lau \u003ckafai@fb.com\u003e\n    Cc: Alexei Starovoitov \u003cast@fb.com\u003e\n    Cc: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Cc: Tejun Heo \u003ctj@kernel.org\u003e\n    Acked-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit a307aca425ed5b6886185d386a85d6cc4a5fb268\nAuthor: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\nDate:   Tue Jun 28 12:18:28 2016 +0200\n\n    bpf: add bpf_skb_change_type helper\n\n    This work adds a helper for changing skb-\u003epkt_type in a controlled way.\n    We only allow a subset of possible values and can extend that in future\n    should other use cases come up. Doing this as a helper has the advantage\n    that errors can be handeled gracefully and thus helper kept extensible.\n\n    It\u0027s a write counterpart to pkt_type member we can already read from\n    struct __sk_buff context. Major use case is to change incoming skbs to\n    PACKET_HOST in a programmatic way instead of having to recirculate via\n    redirect(..., BPF_F_INGRESS), for example.\n\n    Signed-off-by: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Acked-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 33c60bb05b95b72b164b4035a0bb0dfa56e9c8e0\nAuthor: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\nDate:   Tue Jun 28 12:18:27 2016 +0200\n\n    bpf: add bpf_skb_change_proto helper\n\n    This patch adds a minimal helper for doing the groundwork of changing\n    the skb-\u003eprotocol in a controlled way. Currently supported is v4 to\n    v6 and vice versa transitions, which allows f.e. for a minimal, static\n    nat64 implementation where applications in containers that still\n    require IPv4 can be transparently operated in an IPv6-only environment.\n    For example, host facing veth of the container can transparently do\n    the transitions in a programmatic way with the help of clsact qdisc\n    and cls_bpf.\n\n    Idea is to separate concerns for keeping complexity of the helper\n    lower, which means that the programs utilize bpf_skb_change_proto(),\n    bpf_skb_store_bytes() and bpf_lX_csum_replace() to get the job done,\n    instead of doing everything in a single helper (and thus partially\n    duplicating helper functionality). Also, bpf_skb_change_proto()\n    shouldn\u0027t need to deal with raw packet data as this is done by other\n    helpers.\n\n    bpf_skb_proto_6_to_4() and bpf_skb_proto_4_to_6() unclone the skb to\n    operate on a private one, push or pop additionally required header\n    space and migrate the gso/gro meta data from the shared info. We do\n    mark the gso type as dodgy so that headers are checked and segs\n    recalculated by the gso/gro engine. The gso_size target is adapted\n    as well. The flags argument added is currently reserved and can be\n    used for future extensions.\n\n    Signed-off-by: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Acked-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 00df405522dd023da4649327bf4d212bc2137e8e\nAuthor: Martin KaFai Lau \u003ckafai@fb.com\u003e\nDate:   Thu Jun 30 10:28:43 2016 -0700\n\n    cgroup: bpf: Add BPF_MAP_TYPE_CGROUP_ARRAY\n\n    Add a BPF_MAP_TYPE_CGROUP_ARRAY and its bpf_map_ops\u0027s implementations.\n    To update an element, the caller is expected to obtain a cgroup2 backed\n    fd by open(cgroup2_dir) and then update the array with that fd.\n\n    Signed-off-by: Martin KaFai Lau \u003ckafai@fb.com\u003e\n    Cc: Alexei Starovoitov \u003cast@fb.com\u003e\n    Cc: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Cc: Tejun Heo \u003ctj@kernel.org\u003e\n    Acked-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 9717a3efc1196ce2bb34aa6bb6c90fd86df8b14b\nAuthor: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\nDate:   Thu Jun 30 17:24:44 2016 +0200\n\n    bpf: refactor bpf_prog_get and type check into helper\n\n    Since bpf_prog_get() and program type check is used in a couple of places,\n    refactor this into a small helper function that we can make use of. Since\n    the non RO prog-\u003eaux part is not used in performance critical paths and a\n    program destruction via RCU is rather very unlikley when doing the put, we\n    shouldn\u0027t have an issue just doing the bpf_prog_get() + prog-\u003etype !\u003d type\n    check, but actually not taking the ref at all (due to being in fdget() /\n    fdput() section of the bpf fd) is even cleaner and makes the diff smaller\n    as well, so just go for that. Callsites are changed to make use of the new\n    helper where possible.\n\n    Signed-off-by: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Acked-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 9e261a507bb8124fbaf141ab66c7da20ddf52d65\nAuthor: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\nDate:   Thu Jun 30 17:24:43 2016 +0200\n\n    bpf: generally move prog destruction to RCU deferral\n\n    Jann Horn reported following analysis that could potentially result\n    in a very hard to trigger (if not impossible) UAF race, to quote his\n    event timeline:\n\n     - Set up a process with threads T1, T2 and T3\n     - Let T1 set up a socket filter F1 that invokes another filter F2\n       through a BPF map [tail call]\n     - Let T1 trigger the socket filter via a unix domain socket write,\n       don\u0027t wait for completion\n     - Let T2 call PERF_EVENT_IOC_SET_BPF with F2, don\u0027t wait for completion\n     - Now T2 should be behind bpf_prog_get(), but before bpf_prog_put()\n     - Let T3 close the file descriptor for F2, dropping the reference\n       count of F2 to 2\n     - At this point, T1 should have looked up F2 from the map, but not\n       finished executing it\n     - Let T3 remove F2 from the BPF map, dropping the reference count of\n       F2 to 1\n     - Now T2 should call bpf_prog_put() (wrong BPF program type), dropping\n       the reference count of F2 to 0 and scheduling bpf_prog_free_deferred()\n       via schedule_work()\n     - At this point, the BPF program could be freed\n     - BPF execution is still running in a freed BPF program\n\n    While at PERF_EVENT_IOC_SET_BPF time it\u0027s only guaranteed that the perf\n    event fd we\u0027re doing the syscall on doesn\u0027t disappear from underneath us\n    for whole syscall time, it may not be the case for the bpf fd used as\n    an argument only after we did the put. It needs to be a valid fd pointing\n    to a BPF program at the time of the call to make the bpf_prog_get() and\n    while T2 gets preempted, F2 must have dropped reference to 1 on the other\n    CPU. The fput() from the close() in T3 should also add additionally delay\n    to the reference drop via exit_task_work() when bpf_prog_release() gets\n    called as well as scheduling bpf_prog_free_deferred().\n\n    That said, it makes nevertheless sense to move the BPF prog destruction\n    generally after RCU grace period to guarantee that such scenario above,\n    but also others as recently fixed in ceb56070359b (\"bpf, perf: delay release\n    of BPF prog after grace period\") with regards to tail calls won\u0027t happen.\n    Integrating bpf_prog_free_deferred() directly into the RCU callback is\n    not allowed since the invocation might happen from either softirq or\n    process context, so we\u0027re not permitted to block. Reviewing all bpf_prog_put()\n    invocations from eBPF side (note, cBPF -\u003e eBPF progs don\u0027t use this for\n    their destruction) with call_rcu() look good to me.\n\n    Since we don\u0027t know whether at the time of attaching the program, we\u0027re\n    already part of a tail call map, we need to use RCU variant. However, due\n    to this, there won\u0027t be severely more stress on the RCU callback queue:\n    situations with above bpf_prog_get() and bpf_prog_put() combo in practice\n    normally won\u0027t lead to releases, but even if they would, enough effort/\n    cycles have to be put into loading a BPF program into the kernel already.\n\n    Reported-by: Jann Horn \u003cjannh@google.com\u003e\n    Signed-off-by: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Acked-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 18c248c0ff7de1de9df39832a3fbc654d32a7b19\nAuthor: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\nDate:   Tue Jun 28 12:18:26 2016 +0200\n\n    bpf: don\u0027t use raw processor id in generic helper\n\n    Use smp_processor_id() for the generic helper bpf_get_smp_processor_id()\n    instead of the raw variant. This allows for preemption checks when we\n    have DEBUG_PREEMPT, and otherwise uses the raw variant anyway. We only\n    need to keep the raw variant for socket filters, but we can reuse the\n    helper that is already there from cBPF side.\n\n    Signed-off-by: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Acked-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 29b295cc2e5a945f8665a318ddf5c37e9d2cc6eb\nAuthor: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\nDate:   Tue Jun 28 12:18:23 2016 +0200\n\n    bpf: minor cleanups on fd maps and helpers\n\n    Some minor cleanups: i) Remove the unlikely() from fd array map lookups\n    and let the CPU branch predictor do its job, scenarios where there is not\n    always a map entry are very well valid. ii) Move the attribute type check\n    in the bpf_perf_event_read() helper a bit earlier so it\u0027s consistent wrt\n    checks with bpf_perf_event_output() helper as well. iii) remove some\n    comments that are self-documenting in kprobe_prog_is_valid_access() and\n    therefore make it consistent to tp_prog_is_valid_access() as well.\n\n    Signed-off-by: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Acked-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit bf05a1abe8d6579c08f2afe0c9e8ee0b61fa4256\nAuthor: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\nDate:   Wed Jun 15 22:47:14 2016 +0200\n\n    bpf, maps: flush own entries on perf map release\n\n    The behavior of perf event arrays are quite different from all\n    others as they are tightly coupled to perf event fds, f.e. shown\n    recently by commit e03e7ee34fdd (\"perf/bpf: Convert perf_event_array\n    to use struct file\") to make refcounting on perf event more robust.\n    A remaining issue that the current code still has is that since\n    additions to the perf event array take a reference on the struct\n    file via perf_event_get() and are only released via fput() (that\n    cleans up the perf event eventually via perf_event_release_kernel())\n    when the element is either manually removed from the map from user\n    space or automatically when the last reference on the perf event\n    map is dropped. However, this leads us to dangling struct file\u0027s\n    when the map gets pinned after the application owning the perf\n    event descriptor exits, and since the struct file reference will\n    in such case only be manually dropped or via pinned file removal,\n    it leads to the perf event living longer than necessary, consuming\n    needlessly resources for that time.\n\n    Relations between perf event fds and bpf perf event map fds can be\n    rather complex. F.e. maps can act as demuxers among different perf\n    event fds that can possibly be owned by different threads and based\n    on the index selection from the program, events get dispatched to\n    one of the per-cpu fd endpoints. One perf event fd (or, rather a\n    per-cpu set of them) can also live in multiple perf event maps at\n    the same time, listening for events. Also, another requirement is\n    that perf event fds can get closed from application side after they\n    have been attached to the perf event map, so that on exit perf event\n    map will take care of dropping their references eventually. Likewise,\n    when such maps are pinned, the intended behavior is that a user\n    application does bpf_obj_get(), puts its fds in there and on exit\n    when fd is released, they are dropped from the map again, so the map\n    acts rather as connector endpoint. This also makes perf event maps\n    inherently different from program arrays as described in more detail\n    in commit c9da161c6517 (\"bpf: fix clearing on persistent program\n    array maps\").\n\n    To tackle this, map entries are marked by the map struct file that\n    added the element to the map. And when the last reference to that map\n    struct file is released from user space, then the tracked entries\n    are purged from the map. This is okay, because new map struct files\n    instances resp. frontends to the anon inode are provided via\n    bpf_map_new_fd() that is called when we invoke bpf_obj_get_user()\n    for retrieving a pinned map, but also when an initial instance is\n    created via map_create(). The rest is resolved by the vfs layer\n    automatically for us by keeping reference count on the map\u0027s struct\n    file. Any concurrent updates on the map slot are fine as well, it\n    just means that perf_event_fd_array_release() needs to delete less\n    of its own entires.\n\n    Signed-off-by: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Acked-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 67fad7b94bbc2b0b65a4cb90072fc36802d29774\nAuthor: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\nDate:   Wed Jun 15 22:47:13 2016 +0200\n\n    bpf, maps: extend map_fd_get_ptr arguments\n\n    This patch extends map_fd_get_ptr() callback that is used by fd array\n    maps, so that struct file pointer from the related map can be passed\n    in. It\u0027s safe to remove map_update_elem() callback for the two maps since\n    this is only allowed from syscall side, but not from eBPF programs for these\n    two map types. Like in per-cpu map case, bpf_fd_array_map_update_elem()\n    needs to be called directly here due to the extra argument.\n\n    Signed-off-by: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Acked-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 163d05f1b8aac8e276ba238345ea55ed3d02db4a\nAuthor: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\nDate:   Wed Jun 15 22:47:12 2016 +0200\n\n    bpf, maps: add release callback\n\n    Add a release callback for maps that is invoked when the last\n    reference to its struct file is gone and the struct file about\n    to be released by vfs. The handler will be used by fd array maps.\n\n    Signed-off-by: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Acked-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 861cbc2d0d3987683f8276388a51ec750e5aa505\nAuthor: Alexei Starovoitov \u003cast@fb.com\u003e\nDate:   Wed Jun 15 18:25:38 2016 -0700\n\n    bpf: fix matching of data/data_end in verifier\n\n    The ctx structure passed into bpf programs is different depending on bpf\n    program type. The verifier incorrectly marked ctx-\u003edata and ctx-\u003edata_end\n    access based on ctx offset only. That caused loads in tracing programs\n    int bpf_prog(struct pt_regs *ctx) { .. ctx-\u003eax .. }\n    to be incorrectly marked as PTR_TO_PACKET which later caused verifier\n    to reject the program that was actually valid in tracing context.\n    Fix this by doing program type specific matching of ctx offsets.\n\n    Fixes: 969bf05eb3ce (\"bpf: direct packet access\")\n    Reported-by: Sasha Goldshtein \u003cgoldshtn@gmail.com\u003e\n    Signed-off-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Acked-by: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 3d870d75b08b0f778d81af9c1740cf33c7e32830\nAuthor: Arnaldo Carvalho de Melo \u003cacme@redhat.com\u003e\nDate:   Thu Apr 28 13:16:33 2016 -0300\n\n    perf core: Per event callchain limit\n\n    Additionally to being able to control the system wide maximum depth via\n    /proc/sys/kernel/perf_event_max_stack, now we are able to ask for\n    different depths per event, using perf_event_attr.sample_max_stack for\n    that.\n\n    This uses an u16 hole at the end of perf_event_attr, that, when\n    perf_event_attr.sample_type has the PERF_SAMPLE_CALLCHAIN, if\n    sample_max_stack is zero, means use perf_event_max_stack, otherwise\n    it\u0027ll be bounds checked under callchain_mutex.\n\n    Cc: Adrian Hunter \u003cadrian.hunter@intel.com\u003e\n    Cc: Alexander Shishkin \u003calexander.shishkin@linux.intel.com\u003e\n    Cc: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Cc: Brendan Gregg \u003cbrendan.d.gregg@gmail.com\u003e\n    Cc: David Ahern \u003cdsahern@gmail.com\u003e\n    Cc: Frederic Weisbecker \u003cfweisbec@gmail.com\u003e\n    Cc: He Kuang \u003chekuang@huawei.com\u003e\n    Cc: Jiri Olsa \u003cjolsa@redhat.com\u003e\n    Cc: Linus Torvalds \u003ctorvalds@linux-foundation.org\u003e\n    Cc: Masami Hiramatsu \u003cmhiramat@kernel.org\u003e\n    Cc: Milian Wolff \u003cmilian.wolff@kdab.com\u003e\n    Cc: Namhyung Kim \u003cnamhyung@kernel.org\u003e\n    Cc: Peter Zijlstra \u003cpeterz@infradead.org\u003e\n    Cc: Stephane Eranian \u003ceranian@google.com\u003e\n    Cc: Thomas Gleixner \u003ctglx@linutronix.de\u003e\n    Cc: Vince Weaver \u003cvincent.weaver@maine.edu\u003e\n    Cc: Wang Nan \u003cwangnan0@huawei.com\u003e\n    Cc: Zefan Li \u003clizefan@huawei.com\u003e\n    Link: http://lkml.kernel.org/n/tip-kolmn1yo40p7jhswxwrc7rrd@git.kernel.org\n    Signed-off-by: Arnaldo Carvalho de Melo \u003cacme@redhat.com\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit f856f9a7d03bf416794d969d319d763ae4a74166\nAuthor: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\nDate:   Sun May 22 23:16:18 2016 +0200\n\n    bpf, inode: disallow userns mounts\n\n    Follow-up to commit e27f4a942a0e (\"bpf: Use mount_nodev not mount_ns\n    to mount the bpf filesystem\"), which removes the FS_USERNS_MOUNT flag.\n\n    The original idea was to have a per mountns instance instead of a\n    single global fs instance, but that didn\u0027t work out and we had to\n    switch to mount_nodev() model. The intent of that middle ground was\n    that we avoid users who don\u0027t play nice to create endless instances\n    of bpf fs which are difficult to control and discover from an admin\n    point of view, but at the same time it would have allowed us to be\n    more flexible with regard to namespaces.\n\n    Therefore, since we now did the switch to mount_nodev() as a fix\n    where individual instances are created, we also need to remove userns\n    mount flag along with it to avoid running into mentioned situation.\n    I don\u0027t expect any breakage at this early point in time with removing\n    the flag and we can revisit this later should the requirement for\n    this come up with future users. This and commit e27f4a942a0e have\n    been split to facilitate tracking should any of them run into the\n    unlikely case of causing a regression.\n\n    Fixes: b2197755b263 (\"bpf: add support for persistent maps/progs\")\n    Signed-off-by: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Acked-by: Hannes Frederic Sowa \u003channes@stressinduktion.org\u003e\n    Acked-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit f875635e489ea6c4d7519d1bfddabc22c9a7cada\nAuthor: Alexei Starovoitov \u003cast@fb.com\u003e\nDate:   Thu May 19 18:17:14 2016 -0700\n\n    bpf: teach verifier to recognize imm +\u003d ptr pattern\n\n    Humans don\u0027t write C code like:\n      u8 *ptr \u003d skb-\u003edata;\n      int imm \u003d 4;\n      imm +\u003d ptr;\n    but from llvm backend point of view \u0027imm\u0027 and \u0027ptr\u0027 are registers and\n    imm +\u003d ptr may be preferred vs ptr +\u003d imm depending which register value\n    will be used further in the code, while verifier can only recognize ptr +\u003d imm.\n    That caused small unrelated changes in the C code of the bpf program to\n    trigger rejection by the verifier. Therefore teach the verifier to recognize\n    both ptr +\u003d imm and imm +\u003d ptr.\n    For example:\n    when R6\u003dpkt(id\u003d0,off\u003d0,r\u003d62) R7\u003dimm22\n    after r7 +\u003d r6 instruction\n    will be R6\u003dpkt(id\u003d0,off\u003d0,r\u003d62) R7\u003dpkt(id\u003d0,off\u003d22,r\u003d62)\n\n    Fixes: 969bf05eb3ce (\"bpf: direct packet access\")\n    Signed-off-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Acked-by: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 2181f10f7e5bd65c5828617f1c6cc977aac55736\nAuthor: Alexei Starovoitov \u003cast@fb.com\u003e\nDate:   Thu May 19 18:17:13 2016 -0700\n\n    bpf: support decreasing order in direct packet access\n\n    when packet headers are accessed in \u0027decreasing\u0027 order (like TCP port\n    may be fetched before the program reads IP src) the llvm may generate\n    the following code:\n    [...]                // R7\u003dpkt(id\u003d0,off\u003d22,r\u003d70)\n    r2 \u003d *(u32 *)(r7 +0) // good access\n    [...]\n    r7 +\u003d 40             // R7\u003dpkt(id\u003d0,off\u003d62,r\u003d70)\n    r8 \u003d *(u32 *)(r7 +0) // good access\n    [...]\n    r1 \u003d *(u32 *)(r7 -20) // this one will fail though it\u0027s within a safe range\n                          // it\u0027s doing *(u32*)(skb-\u003edata + 42)\n    Fix verifier to recognize such code pattern\n\n    Alos turned out that \u0027off \u003e range\u0027 condition is not a verifier bug.\n    It\u0027s a buggy program that may do something like:\n    if (ptr + 50 \u003e data_end)\n      return 0;\n    ptr +\u003d 60;\n    *(u32*)ptr;\n    in such case emit\n    \"invalid access to packet, off\u003d0 size\u003d4, R1(id\u003d0,off\u003d60,r\u003d50)\" error message,\n    so all information is available for the program author to fix the program.\n\n    Fixes: 969bf05eb3ce (\"bpf: direct packet access\")\n    Signed-off-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Acked-by: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit c29c7a5408970f3d7f223e6d949e3d6b0823c8a6\nAuthor: Eric W. Biederman \u003cebiederm@xmission.com\u003e\nDate:   Fri May 20 17:22:48 2016 -0500\n\n    bpf: Use mount_nodev not mount_ns to mount the bpf filesystem\n\n    While reviewing the filesystems that set FS_USERNS_MOUNT I spotted the\n    bpf filesystem.  Looking at the code I saw a broken usage of mount_ns\n    with current-\u003ensproxy-\u003emnt_ns. As the code does not acquire a\n    reference to the mount namespace it can not possibly be correct to\n    store the mount namespace on the superblock as it does.\n\n    Replace mount_ns with mount_nodev so that each mount of the bpf\n    filesystem returns a distinct instance, and the code is not buggy.\n\n    In discussion with Hannes Frederic Sowa it was reported that the use\n    of mount_ns was an attempt to have one bpf instance per mount\n    namespace, in an attempt to keep resources that pin resources from\n    hiding.  That intent simply does not work, the vfs is not built to\n    allow that kind of behavior.  Which means that the bpf filesystem\n    really is buggy both semantically and in it\u0027s implemenation as it does\n    not nor can it implement the original intent.\n\n    This change is userspace visible, but my experience with similar\n    filesystems leads me to believe nothing will break with a model of each\n    mount of the bpf filesystem is distinct from all others.\n\n    Fixes: b2197755b263 (\"bpf: add support for persistent maps/progs\")\n    Cc: Hannes Frederic Sowa \u003channes@stressinduktion.org\u003e\n    Acked-by: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Signed-off-by: \"Eric W. Biederman\" \u003cebiederm@xmission.com\u003e\n    Acked-by: Hannes Frederic Sowa \u003channes@stressinduktion.org\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 47617cdec8337fd1372a2da5b0eab6ba402e1d7f\nAuthor: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\nDate:   Wed May 18 14:14:28 2016 +0200\n\n    bpf: rather use get_random_int for randomizations\n\n    Start address randomization and blinding in BPF currently use\n    prandom_u32(). prandom_u32() values are not exposed to unpriviledged\n    user space to my knowledge, but given other kernel facilities such as\n    ASLR, stack canaries, etc make use of stronger get_random_int(), we\n    better make use of it here as well given blinding requests successively\n    new random values. get_random_int() has minimal entropy pool depletion,\n    is not cryptographically secure, but doesn\u0027t need to be for our use\n    cases here.\n\n    Suggested-by: Hannes Frederic Sowa \u003channes@stressinduktion.org\u003e\n    Signed-off-by: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Acked-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit d0c23ffd58bb1f3a07a17f305aa02fd0ea7314ca\nAuthor: Arnaldo Carvalho de Melo \u003cacme@redhat.com\u003e\nDate:   Thu Apr 28 12:30:53 2016 -0300\n\n    perf core: Pass max stack as a perf_callchain_entry context\n\n    This makes perf_callchain_{user,kernel}() receive the max stack\n    as context for the perf_callchain_entry, instead of accessing\n    the global sysctl_perf_event_max_stack.\n\n    Cc: Adrian Hunter \u003cadrian.hunter@intel.com\u003e\n    Cc: Alexander Shishkin \u003calexander.shishkin@linux.intel.com\u003e\n    Cc: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Cc: Brendan Gregg \u003cbrendan.d.gregg@gmail.com\u003e\n    Cc: David Ahern \u003cdsahern@gmail.com\u003e\n    Cc: Frederic Weisbecker \u003cfweisbec@gmail.com\u003e\n    Cc: He Kuang \u003chekuang@huawei.com\u003e\n    Cc: Jiri Olsa \u003cjolsa@redhat.com\u003e\n    Cc: Linus Torvalds \u003ctorvalds@linux-foundation.org\u003e\n    Cc: Masami Hiramatsu \u003cmhiramat@kernel.org\u003e\n    Cc: Milian Wolff \u003cmilian.wolff@kdab.com\u003e\n    Cc: Namhyung Kim \u003cnamhyung@kernel.org\u003e\n    Cc: Peter Zijlstra \u003cpeterz@infradead.org\u003e\n    Cc: Stephane Eranian \u003ceranian@google.com\u003e\n    Cc: Thomas Gleixner \u003ctglx@linutronix.de\u003e\n    Cc: Vince Weaver \u003cvincent.weaver@maine.edu\u003e\n    Cc: Wang Nan \u003cwangnan0@huawei.com\u003e\n    Cc: Zefan Li \u003clizefan@huawei.com\u003e\n    Link: http://lkml.kernel.org/n/tip-kolmn1yo40p7jhswxwrc7rrd@git.kernel.org\n    Signed-off-by: Arnaldo Carvalho de Melo \u003cacme@redhat.com\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 1078c71744247854c7c4113949c9f1d30c830209\nAuthor: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\nDate:   Fri May 13 19:08:32 2016 +0200\n\n    bpf: add generic constant blinding for use in jits\n\n    This work adds a generic facility for use from eBPF JIT compilers\n    that allows for further hardening of JIT generated images through\n    blinding constants. In response to the original work on BPF JIT\n    spraying published by Keegan McAllister [1], most BPF JITs were\n    changed to make images read-only and start at a randomized offset\n    in the page, where the rest was filled with trap instructions. We\n    have this nowadays in x86, arm, arm64 and s390 JIT compilers.\n    Additionally, later work also made eBPF interpreter images read\n    only for kernels supporting DEBUG_SET_MODULE_RONX, that is, x86,\n    arm, arm64 and s390 archs as well currently. This is done by\n    default for mentioned JITs when JITing is enabled. Furthermore,\n    we had a generic and configurable constant blinding facility on our\n    todo for quite some time now to further make spraying harder, and\n    first implementation since around netconf 2016.\n\n    We found that for systems where untrusted users can load cBPF/eBPF\n    code where JIT is enabled, start offset randomization helps a bit\n    to make jumps into crafted payload harder, but in case where larger\n    programs that cross page boundary are injected, we again have some\n    part of the program opcodes at a page start offset. With improved\n    guessing and more reliable payload injection, chances can increase\n    to jump into such payload. Elena Reshetova recently wrote a test\n    case for it [2, 3]. Moreover, eBPF comes with 64 bit constants, which\n    can leave some more room for payloads. Note that for all this,\n    additional bugs in the kernel are still required to make the jump\n    (and of course to guess right, to not jump into a trap) and naturally\n    the JIT must be enabled, which is disabled by default.\n\n    For helping mitigation, the general idea is to provide an option\n    bpf_jit_harden that admins can tweak along with bpf_jit_enable, so\n    that for cases where JIT should be enabled for performance reasons,\n    the generated image can be further hardened with blinding constants\n    for unpriviledged users (bpf_jit_harden \u003d\u003d 1), with trading off\n    performance for these, but not for privileged ones. We also added\n    the option of blinding for all users (bpf_jit_harden \u003d\u003d 2), which\n    is quite helpful for testing f.e. with test_bpf.ko. There are no\n    further e.g. hardening levels of bpf_jit_harden switch intended,\n    rationale is to have it dead simple to use as on/off. Since this\n    functionality would need to be duplicated over and over for JIT\n    compilers to use, which are already complex enough, we provide a\n    generic eBPF byte-code level based blinding implementation, which is\n    then just transparently JITed. JIT compilers need to make only a few\n    changes to integrate this facility and can be migrated one by one.\n\n    This option is for eBPF JITs and will be used in x86, arm64, s390\n    without too much effort, and soon ppc64 JITs, thus that native eBPF\n    can be blinded as well as cBPF to eBPF migrations, so that both can\n    be covered with a single implementation. The rule for JITs is that\n    bpf_jit_blind_constants() must be called from bpf_int_jit_compile(),\n    and in case blinding is disabled, we follow normally with JITing the\n    passed program. In case blinding is enabled and we fail during the\n    process of blinding itself, we must return with the interpreter.\n    Similarly, in case the JITing process after the blinding failed, we\n    return normally to the interpreter with the non-blinded code. Meaning,\n    interpreter doesn\u0027t change in any way and operates on eBPF code as\n    usual. For doing this pre-JIT blinding step, we need to make use of\n    a helper/auxiliary register, here BPF_REG_AX. This is strictly internal\n    to the JIT and not in any way part of the eBPF architecture. Just like\n    in the same way as JITs internally make use of some helper registers\n    when emitting code, only that here the helper register is one\n    abstraction level higher in eBPF bytecode, but nevertheless in JIT\n    phase. That helper register is needed since f.e. manually written\n    program can issue loads to all registers of eBPF architecture.\n\n    The core concept with the additional register is: blind out all 32\n    and 64 bit constants by converting BPF_K based instructions into a\n    small sequence from K_VAL into ((RND ^ K_VAL) ^ RND). Therefore, this\n    is transformed into: BPF_REG_AX :\u003d (RND ^ K_VAL), BPF_REG_AX ^\u003d RND,\n    and REG \u003cOP\u003e BPF_REG_AX, so actual operation on the target register\n    is translated from BPF_K into BPF_X one that is operating on\n    BPF_REG_AX\u0027s content. During rewriting phase when blinding, RND is\n    newly generated via prandom_u32() for each processed instruction.\n    64 bit loads are split into two 32 bit loads to make translation and\n    patching not too complex. Only basic thing required by JITs is to\n    call the helper bpf_jit_blind_constants()/bpf_jit_prog_release_other()\n    pair, and to map BPF_REG_AX into an unused register.\n\n    Small bpf_jit_disasm extract from [2] when applied to x86 JIT:\n\n    echo 0 \u003e /proc/sys/net/core/bpf_jit_harden\n\n      ffffffffa034f5e9 + \u003cx\u003e:\n      [...]\n      39:   mov    $0xa8909090,%eax\n      3e:   mov    $0xa8909090,%eax\n      43:   mov    $0xa8ff3148,%eax\n      48:   mov    $0xa89081b4,%eax\n      4d:   mov    $0xa8900bb0,%eax\n      52:   mov    $0xa810e0c1,%eax\n      57:   mov    $0xa8908eb4,%eax\n      5c:   mov    $0xa89020b0,%eax\n      [...]\n\n    echo 1 \u003e /proc/sys/net/core/bpf_jit_harden\n\n      ffffffffa034f1e5 + \u003cx\u003e:\n      [...]\n      39:   mov    $0xe1192563,%r10d\n      3f:   xor    $0x4989b5f3,%r10d\n      46:   mov    %r10d,%eax\n      49:   mov    $0xb8296d93,%r10d\n      4f:   xor    $0x10b9fd03,%r10d\n      56:   mov    %r10d,%eax\n      59:   mov    $0x8c381146,%r10d\n      5f:   xor    $0x24c7200e,%r10d\n      66:   mov    %r10d,%eax\n      69:   mov    $0xeb2a830e,%r10d\n      6f:   xor    $0x43ba02ba,%r10d\n      76:   mov    %r10d,%eax\n      79:   mov    $0xd9730af,%r10d\n      7f:   xor    $0xa5073b1f,%r10d\n      86:   mov    %r10d,%eax\n      89:   mov    $0x9a45662b,%r10d\n      8f:   xor    $0x325586ea,%r10d\n      96:   mov    %r10d,%eax\n      [...]\n\n    As can be seen, original constants that carry payload are hidden\n    when enabled, actual operations are transformed from constant-based\n    to register-based ones, making jumps into constants ineffective.\n    Above extract/example uses single BPF load instruction over and\n    over, but of course all instructions with constants are blinded.\n\n    Performance wise, JIT with blinding performs a bit slower than just\n    JIT and faster than interpreter case. This is expected, since we\n    still get all the performance benefits from JITing and in normal\n    use-cases not every single instruction needs to be blinded. Summing\n    up all 296 test cases averaged over multiple runs from test_bpf.ko\n    suite, interpreter was 55% slower than JIT only and JIT with blinding\n    was 8% slower than JIT only. Since there are also some extremes in\n    the test suite, I expect for ordinary workloads that the performance\n    for the JIT with blinding case is even closer to JIT only case,\n    f.e. nmap test case from suite has averaged timings in ns 29 (JIT),\n    35 (+ blinding), and 151 (interpreter).\n\n    BPF test suite, seccomp test suite, eBPF sample code and various\n    bigger networking eBPF programs have been tested with this and were\n    running fine. For testing purposes, I also adapted interpreter and\n    redirected blinded eBPF image to interpreter and also here all tests\n    pass.\n\n      [1] http://mainisusuallyafunction.blogspot.com/2012/11/attacking-hardened-linux-systems-with.html\n      [2] https://github.com/01org/jit-spray-poc-for-ksp/\n      [3] http://www.openwall.com/lists/kernel-hardening/2016/05/03/5\n\n    Signed-off-by: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Reviewed-by: Elena Reshetova \u003celena.reshetova@intel.com\u003e\n    Acked-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit aafe2056243377c1af840c886a914d39e9e8c6db\nAuthor: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\nDate:   Fri May 13 19:08:27 2016 +0200\n\n    bpf: move bpf_jit_enable declaration\n\n    Move the bpf_jit_enable declaration to the filter.h file where\n    most other core code is declared, also since we\u0027re going to add\n    a second knob there.\n\n    Signed-off-by: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Acked-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit b11b43db4b3336724bb40ac5c52a292862f65b65\nAuthor: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\nDate:   Sat Jun 4 20:50:59 2016 +0200\n\n    bpf, trace: use READ_ONCE for retrieving file ptr\n\n    In bpf_perf_event_read() and bpf_perf_event_output(), we must use\n    READ_ONCE() for fetching the struct file pointer, which could get\n    updated concurrently, so we must prevent the compiler from potential\n    refetching.\n\n    We already do this with tail calls for fetching the related bpf_prog,\n    but not so on stored perf events. Semantics for both are the same\n    with regards to updates.\n\n    Fixes: a43eec304259 (\"bpf: introduce bpf_perf_event_output() helper\")\n    Fixes: 35578d798400 (\"bpf: Implement function bpf_perf_event_read() that get the selected hardware PMU conuter\")\n    Signed-off-by: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Acked-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 7833af0f20ddc7ce7f7ce93aa7e1baecd4afb0fc\nAuthor: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\nDate:   Mon Apr 18 21:01:23 2016 +0200\n\n    bpf, trace: add BPF_F_CURRENT_CPU flag for bpf_perf_event_output\n\n    Add a BPF_F_CURRENT_CPU flag to optimize the use-case where user space has\n    per-CPU ring buffers and the eBPF program pushes the data into the current\n    CPU\u0027s ring buffer which saves us an extra helper function call in eBPF.\n    Also, make sure to properly reserve the remaining flags which are not used.\n\n    Signed-off-by: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Signed-off-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 2322bf96f733600232f2c4278b130e1d7c19f491\nAuthor: Arnd Bergmann \u003carnd@arndb.de\u003e\nDate:   Sat Apr 16 22:29:33 2016 +0200\n\n    bpf: avoid warning for wrong pointer cast\n\n    Two new functions in bpf contain a cast from a \u0027u64\u0027 to a\n    pointer. This works on 64-bit architectures but causes a warning\n    on all 32-bit architectures:\n\n    kernel/trace/bpf_trace.c: In function \u0027bpf_perf_event_output_tp\u0027:\n    kernel/trace/bpf_trace.c:350:13: error: cast to pointer from integer of different size [-Werror\u003dint-to-pointer-cast]\n      u64 ctx \u003d *(long *)r1;\n\n    This changes the cast to first convert the u64 argument into a uintptr_t,\n    which is guaranteed to be the same size as a pointer.\n\n    Signed-off-by: Arnd Bergmann \u003carnd@arndb.de\u003e\n    Fixes: 9940d67c93b5 (\"bpf: support bpf_get_stackid() and bpf_perf_event_output() in tracepoint programs\")\n    Acked-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 5349ef5f720bc92e473ccde5ca49b2ee4ba878cb\nAuthor: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\nDate:   Fri May 13 19:08:31 2016 +0200\n\n    bpf: prepare bpf_int_jit_compile/bpf_prog_select_runtime apis\n\n    Since the blinding is strictly only called from inside eBPF JITs,\n    we need to change signatures for bpf_int_jit_compile() and\n    bpf_prog_select_runtime() first in order to prepare that the\n    eBPF program we\u0027re dealing with can change underneath. Hence,\n    for call sites, we need to return the latest prog. No functional\n    change in this patch.\n\n    Signed-off-by: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Acked-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 192e276c6a5071da00428fc6b5cde067c5c01702\nAuthor: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\nDate:   Fri May 13 19:08:30 2016 +0200\n\n    bpf: add bpf_patch_insn_single helper\n\n    Move the functionality to patch instructions out of the verifier\n    code and into the core as the new bpf_patch_insn_single() helper\n    will be needed later on for blinding as well. No changes in\n    functionality.\n\n    Signed-off-by: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Acked-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit dcd52ad90b5721da28528f807e77dd0dc3d7962f\nAuthor: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\nDate:   Fri May 13 19:08:26 2016 +0200\n\n    bpf: minor cleanups in ebpf code\n\n    Besides others, remove redundant comments where the code is self\n    documenting enough, and properly indent various bpf_verifier_ops\n    and bpf_prog_type_list declarations. Moreover, remove two exports\n    that actually have no module user.\n\n    Signed-off-by: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Acked-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit d4e7499c984128d8509f5ac3bf647f3295d84c4a\nAuthor: Alexei Starovoitov \u003cast@fb.com\u003e\nDate:   Thu May 5 19:49:11 2016 -0700\n\n    bpf: improve verifier state equivalence\n\n    since UNKNOWN_VALUE type is weaker than CONST_IMM we can un-teach\n    verifier its recognition of constants in conditional branches\n    without affecting safety.\n    Ex:\n    if (reg \u003d\u003d 123) {\n      .. here verifier was marking reg-\u003etype as CONST_IMM\n         instead keep reg as UNKNOWN_VALUE\n    }\n\n    Two verifier states with UNKNOWN_VALUE are equivalent, whereas\n    CONST_IMM_X !\u003d CONST_IMM_Y, since CONST_IMM is used for stack range\n    verification and other cases.\n    So help search pruning by marking registers as UNKNOWN_VALUE\n    where possible instead of CONST_IMM.\n\n    Signed-off-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Acked-by: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit a74ad8ba8c5203757e5c0b7d445dc7224df4b18e\nAuthor: Alexei Starovoitov \u003cast@fb.com\u003e\nDate:   Thu May 5 19:49:10 2016 -0700\n\n    bpf: direct packet access\n\n    Extended BPF carried over two instructions from classic to access\n    packet data: LD_ABS and LD_IND. They\u0027re highly optimized in JITs,\n    but due to their design they have to do length check for every access.\n    When BPF is processing 20M packets per second single LD_ABS after JIT\n    is consuming 3% cpu. Hence the need to optimize it further by amortizing\n    the cost of \u0027off \u003c skb_headlen\u0027 over multiple packet accesses.\n    One option is to introduce two new eBPF instructions LD_ABS_DW and LD_IND_DW\n    with similar usage as skb_header_pointer().\n    The kernel part for interpreter and x64 JIT was implemented in [1], but such\n    new insns behave like old ld_abs and abort the program with \u0027return 0\u0027 if\n    access is beyond linear data. Such hidden control flow is hard to workaround\n    plus changing JITs and rolling out new llvm is incovenient.\n\n    Therefore allow cls_bpf/act_bpf program access skb-\u003edata directly:\n    int bpf_prog(struct __sk_buff *skb)\n    {\n      struct iphdr *ip;\n\n      if (skb-\u003edata + sizeof(struct iphdr) + ETH_HLEN \u003e skb-\u003edata_end)\n          /* packet too small */\n          return 0;\n\n      ip \u003d skb-\u003edata + ETH_HLEN;\n\n      /* access IP header fields with direct loads */\n      if (ip-\u003eversion !\u003d 4 || ip-\u003esaddr \u003d\u003d 0x7f000001)\n          return 1;\n      [...]\n    }\n\n    This solution avoids introduction of new instructions. llvm stays\n    the same and all JITs stay the same, but verifier has to work extra hard\n    to prove safety of the above program.\n\n    For XDP the direct store instructions can be allowed as well.\n\n    The skb-\u003edata is NET_IP_ALIGNED, so for common cases the verifier can check\n    the alignment. The complex packet parsers where packet pointer is adjusted\n    incrementally cannot be tracked for alignment, so allow byte access in such cases\n    and misaligned access on architectures that define efficient_unaligned_access\n\n    [1] https://git.kernel.org/cgit/linux/kernel/git/ast/bpf.git/?h\u003dld_abs_dw\n\n    Signed-off-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Acked-by: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 80227b6f68c8f8b482fd91a972b3e63bd51f3f2d\nAuthor: Alexei Starovoitov \u003cast@fb.com\u003e\nDate:   Thu May 5 19:49:09 2016 -0700\n\n    bpf: cleanup verifier code\n\n    cleanup verifier code and prepare it for addition of \"pointer to packet\" logic\n\n    Signed-off-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Acked-by: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 7d51802990f19293d3019010dab50455612c3ddd\nAuthor: Alexei Starovoitov \u003cast@fb.com\u003e\nDate:   Wed Apr 27 18:56:21 2016 -0700\n\n    bpf: fix check_map_func_compatibility logic\n\n    The commit 35578d798400 (\"bpf: Implement function bpf_perf_event_read() that get the selected hardware PMU conuter\")\n    introduced clever way to check bpf_helper\u003c-\u003emap_type compatibility.\n    Later on commit a43eec304259 (\"bpf: introduce bpf_perf_event_output() helper\") adjusted\n    the logic and inadvertently broke it.\n    Get rid of the clever bool compare and go back to two-way check\n    from map and from helper perspective.\n\n    Fixes: a43eec304259 (\"bpf: introduce bpf_perf_event_output() helper\")\n    Reported-by: Jann Horn \u003cjannh@google.com\u003e\n    Signed-off-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Signed-off-by: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 58931fc4de12dd0b23f0b0bc8609b2bdbf519757\nAuthor: Alexei Starovoitov \u003cast@fb.com\u003e\nDate:   Wed Apr 27 18:56:20 2016 -0700\n\n    bpf: fix refcnt overflow\n\n    On a system with \u003e32Gbyte of phyiscal memory and infinite RLIMIT_MEMLOCK,\n    the malicious application may overflow 32-bit bpf program refcnt.\n    It\u0027s also possible to overflow map refcnt on 1Tb system.\n    Impose 32k hard limit which means that the same bpf program or\n    map cannot be shared by more than 32k processes.\n\n    Fixes: 1be7f75d1668 (\"bpf: enable non-root eBPF programs\")\n    Reported-by: Jann Horn \u003cjannh@google.com\u003e\n    Signed-off-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Acked-by: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 3f4206f8f81aa8a5c0f7e96ea8fd656bdddb524e\nAuthor: Arnaldo Carvalho de Melo \u003cacme@redhat.com\u003e\nDate:   Thu Apr 21 12:28:50 2016 -0300\n\n    perf core: Allow setting up max frame stack depth via sysctl\n\n    The default remains 127, which is good for most cases, and not even hit\n    most of the time, but then for some cases, as reported by Brendan, 1024+\n    deep frames are appearing on the radar for things like groovy, ruby.\n\n    And in some workloads putting a _lower_ cap on this may make sense. One\n    that is per event still needs to be put in place tho.\n\n    The new file is:\n\n      # cat /proc/sys/kernel/perf_event_max_stack\n      127\n\n    Chaging it:\n\n      # echo 256 \u003e /proc/sys/kernel/perf_event_max_stack\n      # cat /proc/sys/kernel/perf_event_max_stack\n      256\n\n    But as soon as there is some event using callchains we get:\n\n      # echo 512 \u003e /proc/sys/kernel/perf_event_max_stack\n      -bash: echo: write error: Device or resource busy\n      #\n\n    Because we only allocate the callchain percpu data structures when there\n    is a user, which allows for changing the max easily, its just a matter\n    of having no callchain users at that point.\n\n    Reported-and-Tested-by: Brendan Gregg \u003cbrendan.d.gregg@gmail.com\u003e\n    Reviewed-by: Frederic Weisbecker \u003cfweisbec@gmail.com\u003e\n    Acked-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Acked-by: David Ahern \u003cdsahern@gmail.com\u003e\n    Cc: Adrian Hunter \u003cadrian.hunter@intel.com\u003e\n    Cc: Alexander Shishkin \u003calexander.shishkin@linux.intel.com\u003e\n    Cc: He Kuang \u003chekuang@huawei.com\u003e\n    Cc: Jiri Olsa \u003cjolsa@redhat.com\u003e\n    Cc: Linus Torvalds \u003ctorvalds@linux-foundation.org\u003e\n    Cc: Masami Hiramatsu \u003cmhiramat@kernel.org\u003e\n    Cc: Milian Wolff \u003cmilian.wolff@kdab.com\u003e\n    Cc: Namhyung Kim \u003cnamhyung@kernel.org\u003e\n    Cc: Peter Zijlstra \u003cpeterz@infradead.org\u003e\n    Cc: Stephane Eranian \u003ceranian@google.com\u003e\n    Cc: Thomas Gleixner \u003ctglx@linutronix.de\u003e\n    Cc: Vince Weaver \u003cvincent.weaver@maine.edu\u003e\n    Cc: Wang Nan \u003cwangnan0@huawei.com\u003e\n    Cc: Zefan Li \u003clizefan@huawei.com\u003e\n    Link: http://lkml.kernel.org/r/20160426002928.GB16708@kernel.org\n    Signed-off-by: Arnaldo Carvalho de Melo \u003cacme@redhat.com\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 055afa297fdfd4beb890a1cbd8428cd01e9953c5\nAuthor: Alexei Starovoitov \u003cast@fb.com\u003e\nDate:   Wed Feb 17 19:58:57 2016 -0800\n\n    perf: generalize perf_callchain\n\n    . avoid walking the stack when there is no room left in the buffer\n    . generalize get_perf_callchain() to be called from bpf helper\n\n    Signed-off-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit a3ea6d47cc0eb62ce408ed6d4353abc5739266f4\nAuthor: Jann Horn \u003cjannh@google.com\u003e\nDate:   Tue Apr 26 22:26:26 2016 +0200\n\n    bpf: fix double-fdput in replace_map_fd_with_map_ptr()\n\n    When bpf(BPF_PROG_LOAD, ...) was invoked with a BPF program whose bytecode\n    references a non-map file descriptor as a map file descriptor, the error\n    handling code called fdput() twice instead of once (in __bpf_map_get() and\n    in replace_map_fd_with_map_ptr()). If the file descriptor table of the\n    current task is shared, this causes f_count to be decremented too much,\n    allowing the struct file to be freed while it is still in use\n    (use-after-free). This can be exploited to gain root privileges by an\n    unprivileged user.\n\n    This bug was introduced in\n    commit 0246e64d9a5f (\"bpf: handle pseudo BPF_LD_IMM64 insn\"), but is only\n    exploitable since\n    commit 1be7f75d1668 (\"bpf: enable non-root eBPF programs\") because\n    previously, CAP_SYS_ADMIN was required to reach the vulnerable code.\n\n    (posted publicly according to request by maintainer)\n\n    Signed-off-by: Jann Horn \u003cjannh@google.com\u003e\n    Signed-off-by: Linus Torvalds \u003ctorvalds@linux-foundation.org\u003e\n    Acked-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Acked-by: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 4a8ab235c7c50b3000e388aa86738d16a1cf9224\nAuthor: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\nDate:   Mon Apr 18 21:01:24 2016 +0200\n\n    bpf: add event output helper for notifications/sampling/logging\n\n    This patch adds a new helper for cls/act programs that can push events\n    to user space applications. For networking, this can be f.e. for sampling,\n    debugging, logging purposes or pushing of arbitrary wake-up events. The\n    idea is similar to a43eec304259 (\"bpf: introduce bpf_perf_event_output()\n    helper\") and 39111695b1b8 (\"samples: bpf: add bpf_perf_event_output example\").\n\n    The eBPF program utilizes a perf event array map that user space populates\n    with fds from perf_event_open(), the eBPF program calls into the helper\n    f.e. as skb_event_output(skb, \u0026my_map, BPF_F_CURRENT_CPU, raw, sizeof(raw))\n    so that the raw data is pushed into the fd f.e. at the map index of the\n    current CPU.\n\n    User space can poll/mmap/etc on this and has a data channel for receiving\n    events that can be post-processed. The nice thing is that since the eBPF\n    program and user space application making use of it are tightly coupled,\n    they can define their own arbitrary raw data format and what/when they\n    want to push.\n\n    While f.e. packet headers could be one part of the meta data that is being\n    pushed, this is not a substitute for things like packet sockets as whole\n    packet is not being pushed and push is only done in a single direction.\n    Intention is more of a generically usable, efficient event pipe to applications.\n    Workflow is that tc can pin the map and applications can attach themselves\n    e.g. after cls/act setup to one or multiple map slots, demuxing is done by\n    the eBPF program.\n\n    Adding this facility is with minimal effort, it reuses the helper\n    introduced in a43eec304259 (\"bpf: introduce bpf_perf_event_output() helper\")\n    and we get its functionality for free by overloading its BPF_FUNC_ identifier\n    for cls/act programs, ctx is currently unused, but will be made use of in\n    future. Example will be added to iproute2\u0027s BPF example files.\n\n    Signed-off-by: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Signed-off-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 1e65fa3c7d3cddc9e164ccf92b7bc2f58af295ec\nAuthor: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\nDate:   Wed Apr 13 00:10:52 2016 +0200\n\n    bpf: convert relevant helper args to ARG_PTR_TO_RAW_STACK\n\n    This patch converts all helpers that can use ARG_PTR_TO_RAW_STACK as argument\n    type. For tc programs this is bpf_skb_load_bytes(), bpf_skb_get_tunnel_key(),\n    bpf_skb_get_tunnel_opt(). For tracing, this optimizes bpf_get_current_comm()\n    and bpf_probe_read(). The check in bpf_skb_load_bytes() for MAX_BPF_STACK can\n    also be removed since the verifier already makes sure we stay within bounds\n    on stack buffers.\n\n    Signed-off-by: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Acked-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit ae0139946a0417c15215f615e076d2435cc14259\nAuthor: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\nDate:   Wed Mar 30 00:02:00 2016 +0200\n\n    bpf: make padding in bpf_tunnel_key explicit\n\n    Make the 2 byte padding in struct bpf_tunnel_key between tunnel_ttl\n    and tunnel_label members explicit. No issue has been observed, and\n    gcc/llvm does padding for the old struct already, where tunnel_label\n    was not yet present, so the current code works, but since it\u0027s part\n    of uapi, make sure we don\u0027t introduce holes in structs.\n\n    Therefore, add tunnel_ext that we can use generically in future\n    (f.e. to flag OAM messages for backends, etc). Also add the offset\n    to the compat tests to be sure should some compilers not padd the\n    tail of the old version of bpf_tunnel_key.\n\n    Fixes: 4018ab1875e0 (\"bpf: support flow label for bpf_skb_{set, get}_tunnel_key\")\n    Signed-off-by: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Acked-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 3bcd51201b2497fa46a6bad2b0b0ddd47ae7061a\nAuthor: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\nDate:   Wed Mar 16 01:42:51 2016 +0100\n\n    ip_tunnels, bpf: define IP_TUNNEL_OPTS_MAX and use it\n\n    eBPF defines this as BPF_TUNLEN_MAX and OVS just uses the hard-coded\n    value inside struct sw_flow_key. Thus, add and use IP_TUNNEL_OPTS_MAX\n    for this, which makes the code a bit more generic and allows to remove\n    BPF_TUNLEN_MAX from eBPF code.\n\n    Signed-off-by: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit b20ac310bea9199559b104448f38c9c3a088d5b5\nAuthor: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\nDate:   Wed Mar 16 01:42:50 2016 +0100\n\n    bpf, dst: add and use dst_tclassid helper\n\n    We can just add a small helper dst_tclassid() for retrieving the\n    dst-\u003etclassid value. It makes the code a bit better in that we can\n    get rid of the ifdef from filter.c by moving this into the header.\n\n    Signed-off-by: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 3f13d80149ee747b142959233b94fb5ba0b21f23\nAuthor: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\nDate:   Wed Mar 16 01:42:49 2016 +0100\n\n    bpf: make skb-\u003etc_classid also readable\n\n    Currently, the tc_classid from eBPF skb context is write-only, but there\u0027s\n    no good reason for tc programs to limit it to write-only. For example,\n    it can be used to transfer its state via tail calls where the resulting\n    tc_classid gets filled gradually.\n\n    Signed-off-by: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Acked-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit b3b2fa8199462acadf80fdcbfba82bb5d480a76f\nAuthor: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\nDate:   Wed Mar 9 03:00:05 2016 +0100\n\n    bpf: support flow label for bpf_skb_{set, get}_tunnel_key\n\n    This patch extends bpf_tunnel_key with a tunnel_label member, that maps\n    to ip_tunnel_key\u0027s label so underlying backends like vxlan and geneve\n    can propagate the label to udp_tunnel6_xmit_skb(), where it\u0027s being set\n    in the IPv6 header. It allows for having 20 more bits to encode/decode\n    flow related meta information programmatically. Tested with vxlan and\n    geneve.\n\n    Signed-off-by: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Acked-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 802357ddae788cdbcd3ad37bddaa1bb5bc68c36c\nAuthor: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\nDate:   Fri Mar 4 15:15:06 2016 +0100\n\n    bpf: support for access to tunnel options\n\n    After eBPF being able to programmatically access/manage tunnel key meta\n    data via commit d3aa45ce6b94 (\"bpf: add helpers to access tunnel metadata\")\n    and more recently also for IPv6 through c6c33454072f (\"bpf: support ipv6\n    for bpf_skb_{set,get}_tunnel_key\"), this work adds two complementary\n    helpers to generically access their auxiliary tunnel options.\n\n    Geneve and vxlan support this facility. For geneve, TLVs can be pushed,\n    and for the vxlan case its GBP extension. I.e. setting tunnel key for geneve\n    case only makes sense, if we can also read/write TLVs into it. In the GBP\n    case, it provides the flexibility to easily map the group policy ID in\n    combination with other helpers or maps.\n\n    I chose to model this as two separate helpers, bpf_skb_{set,get}_tunnel_opt(),\n    for a couple of reasons. bpf_skb_{set,get}_tunnel_key() is already rather\n    complex by itself, and there may be cases for tunnel key backends where\n    tunnel options are not always needed. If we would have integrated this\n    into bpf_skb_{set,get}_tunnel_key() nevertheless, we are very limited with\n    remaining helper arguments, so keeping compatibility on structs in case of\n    passing in a flat buffer gets more cumbersome. Separating both also allows\n    for more flexibility and future extensibility, f.e. options could be fed\n    directly from a map, etc.\n\n    Moreover, change geneve\u0027s xmit path to test only for info-\u003eoptions_len\n    instead of TUNNEL_GENEVE_OPT flag. This makes it more consistent with vxlan\u0027s\n    xmit path and allows for avoiding to specify a protocol flag in the API on\n    xmit, so it can be protocol agnostic. Having info-\u003eoptions_len is enough\n    information that is needed. Tested with vxlan and geneve.\n\n    Signed-off-by: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Acked-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit fe0fb5c15b4f9450ced349b014bbb959fdf061db\nAuthor: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\nDate:   Fri Mar 4 15:15:05 2016 +0100\n\n    bpf: allow to propagate df in bpf_skb_set_tunnel_key\n\n    Added by 9a628224a61b (\"ip_tunnel: Add dont fragment flag.\"), allow to\n    feed df flag into tunneling facilities (currently supported on TX by\n    vxlan, geneve and gre) as a hint from eBPF\u0027s bpf_skb_set_tunnel_key()\n    helper.\n\n    Signed-off-by: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Acked-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 6d03106a7d802d3c7bf9db8663855e2e546827d4\nAuthor: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\nDate:   Fri Mar 4 15:15:04 2016 +0100\n\n    bpf: make helper function protos static\n\n    They are only used here, so there\u0027s no reason they should not be static.\n    Only the vlan push/pop protos are used in the test_bpf suite.\n\n    Signed-off-by: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Acked-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 626cceff94ac482ea00425dbfc264911a122215d\nAuthor: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\nDate:   Fri Mar 4 15:15:03 2016 +0100\n\n    bpf: add flags to bpf_skb_store_bytes for clearing hash\n\n    When overwriting parts of the packet with bpf_skb_store_bytes() that\n    were fed previously into skb-\u003ehash calculation, we should clear the\n    current hash with skb_clear_hash(), so that a next skb_get_hash() call\n    can determine the correct hash related to this skb.\n\n    Signed-off-by: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Acked-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit fdf2f59cd69829c41498e9ec68dd56d0faf15e27\nAuthor: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\nDate:   Fri Mar 4 15:15:02 2016 +0100\n\n    bpf: allow bpf_csum_diff to feed bpf_l3_csum_replace as well\n\n    Commit 7d672345ed29 (\"bpf: add generic bpf_csum_diff helper\") added a\n    generic checksum diff helper that can feed bpf_l4_csum_replace() with\n    a target __wsum diff that is to be applied to the L4 checksum. This\n    facility is very flexible, can be cascaded, allows for adding, removing,\n    or diffing data, or for calculating the pseudo header checksum from\n    scratch, but it can also be reused for working with the IPv4 header\n    checksum.\n\n    Thus, analogous to bpf_l4_csum_replace(), add a case for header field\n    value of 0 to change the checksum at a given offset through a new helper\n    csum_replace_by_diff(). Also, in addition to that, this provides an\n    easy to use interface for feeding precalculated diffs f.e. coming from\n    a map. It nicely complements bpf_l3_csum_replace() that currently allows\n    only for csum updates of 2 and 4 byte diffs.\n\n    Signed-off-by: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Acked-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 648d4535f4b232f73d4eed3e3bd24a30d4ab4462\nAuthor: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\nDate:   Tue Feb 23 02:05:26 2016 +0100\n\n    bpf: fix csum setting for bpf_set_tunnel_key\n\n    The fix in 35e2d1152b22 (\"tunnels: Allow IPv6 UDP checksums to be correctly\n    controlled.\") changed behavior for bpf_set_tunnel_key() when in use with\n    IPv6 and thus uncovered a bug that TUNNEL_CSUM needed to be set but wasn\u0027t.\n    As a result, the stack dropped ingress vxlan IPv6 packets, that have been\n    sent via eBPF through collect meta data mode due to checksum now being zero.\n\n    Since after LCO, we enable IPv4 checksum by default, so make that analogous\n    and only provide a flag BPF_F_ZERO_CSUM_TX for the user to turn it off in\n    IPv4 case.\n\n    Fixes: 35e2d1152b22 (\"tunnels: Allow IPv6 UDP checksums to be correctly controlled.\")\n    Fixes: c6c33454072f (\"bpf: support ipv6 for bpf_skb_{set,get}_tunnel_key\")\n    Signed-off-by: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Acked-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit ed44455580e72cc8c9f31cbc3c5d10cc4899302e\nAuthor: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\nDate:   Fri Feb 19 23:05:26 2016 +0100\n\n    bpf: fix csum update in bpf_l4_csum_replace helper for udp\n\n    When using this helper for updating UDP checksums, we need to extend\n    this in order to write CSUM_MANGLED_0 for csum computations that result\n    into 0 as sum. Reason we need this is because packets with a checksum\n    could otherwise become incorrectly marked as a packet without a checksum.\n    Likewise, if the user indicates BPF_F_MARK_MANGLED_0, then we should\n    not turn packets without a checksum into ones with a checksum.\n\n    Signed-off-by: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Acked-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit aa4e2e4ae0d05d08035d8d4b64e7a8f3cacb7c06\nAuthor: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\nDate:   Fri Feb 19 23:05:27 2016 +0100\n\n    bpf: don\u0027t emit mov A,A on return\n\n    While debugging with bpf_jit_disasm I noticed emissions of \u0027mov %eax,%eax\u0027,\n    and found that this comes from BPF_RET | BPF_A translations from classic\n    BPF. Emitting this is unnecessary as BPF_REG_A is mapped into BPF_REG_0\n    already, therefore only emit a mov when immediates are used as return value.\n\n    Signed-off-by: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Acked-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 82d73e7a1eeec0d35c441677415876f2a049fadb\nAuthor: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\nDate:   Fri Feb 19 23:05:24 2016 +0100\n\n    bpf: remove artificial bpf_skb_{load, store}_bytes buffer limitation\n\n    We currently limit bpf_skb_store_bytes() and bpf_skb_load_bytes()\n    helpers to only store or load a maximum buffer of 16 bytes. Thus,\n    loading, rewriting and storing headers require several bpf_skb_load_bytes()\n    and bpf_skb_store_bytes() calls.\n\n    Also here we can use a per-cpu scratch buffer instead in order to not\n    pressure stack space any further. I do suspect that this limit was mainly\n    set in place for this particular reason. So, ease program development\n    by removing this limitation and make the scratchpad generic, so it can\n    be reused.\n\n    Signed-off-by: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Acked-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit f35e92fc2d57591123bad96b6fd88745d56639ca\nAuthor: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\nDate:   Fri Feb 19 23:05:23 2016 +0100\n\n    bpf: add generic bpf_csum_diff helper\n\n    For L4 checksums, we currently have bpf_l4_csum_replace() helper. It\u0027s\n    currently limited to handle 2 and 4 byte changes in a header and feeds the\n    from/to into inet_proto_csum_replace{2,4}() helpers of the kernel. When\n    working with IPv6, for example, this makes it rather cumbersome to deal\n    with, similarly when editing larger parts of a header.\n\n    Instead, extend the API in a more generic way: For bpf_l4_csum_replace(),\n    add a case for header field mask of 0 to change the checksum at a given\n    offset through inet_proto_csum_replace_by_diff(), and provide a helper\n    bpf_csum_diff() that can generically calculate a from/to diff for arbitrary\n    amounts of data.\n\n    This can be used in multiple ways: for the bpf_l4_csum_replace() only\n    part, this even provides us with the option to insert precalculated diffs\n    from user space f.e. from a map, or from bpf_csum_diff() during runtime.\n\n    bpf_csum_diff() has a optional from/to stack buffer input, so we can\n    calculate a diff by using a scratchbuffer for scenarios where we\u0027re\n    inserting (from is NULL), removing (to is NULL) or diffing (from/to buffers\n    don\u0027t need to be of equal size) data. Also, bpf_csum_diff() allows to\n    feed a previous csum into csum_partial(), so the function can also be\n    cascaded.\n\n    Signed-off-by: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Acked-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 9a9393327c7773ddf574cdb94c920fe246d0b169\nAuthor: Hannes Frederic Sowa \u003channes@stressinduktion.org\u003e\nDate:   Tue Apr 5 17:10:16 2016 +0200\n\n    tun: use socket locks for sk_{attach,detatch}_filter\n\n    This reverts commit 5a5abb1fa3b05dd (\"tun, bpf: fix suspicious RCU usage\n    in tun_{attach, detach}_filter\") and replaces it to use lock_sock around\n    sk_{attach,detach}_filter. The checks inside filter.c are updated with\n    lockdep_sock_is_held to check for proper socket locks.\n\n    It keeps the code cleaner by ensuring that only one lock governs the\n    socket filter instead of two independent locks.\n\n    Cc: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Signed-off-by: Hannes Frederic Sowa \u003channes@stressinduktion.org\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 5543a384f2aa671e9aeb83de915cb685517db26c\nAuthor: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\nDate:   Wed Apr 13 00:10:51 2016 +0200\n\n    bpf, verifier: add ARG_PTR_TO_RAW_STACK type\n\n    When passing buffers from eBPF stack space into a helper function, we have\n    ARG_PTR_TO_STACK argument type for helpers available. The verifier makes sure\n    that such buffers are initialized, within boundaries, etc.\n\n    However, the downside with this is that we have a couple of helper functions\n    such as bpf_skb_load_bytes() that fill out the passed buffer in the expected\n    success case anyway, so zero initializing them prior to the helper call is\n    unneeded/wasted instructions in the eBPF program that can be avoided.\n\n    Therefore, add a new helper function argument type called ARG_PTR_TO_RAW_STACK.\n    The idea is to skip the STACK_MISC check in check_stack_boundary() and color\n    the related stack slots as STACK_MISC after we checked all call arguments.\n\n    Helper functions using ARG_PTR_TO_RAW_STACK must make sure that every path of\n    the helper function will fill the provided buffer area, so that we cannot leak\n    any uninitialized stack memory. This f.e. means that error paths need to\n    memset() the buffers, but the expected fast-path doesn\u0027t have to do this\n    anymore.\n\n    Since there\u0027s no such helper needing more than at most one ARG_PTR_TO_RAW_STACK\n    argument, we can keep it simple and don\u0027t need to check for multiple areas.\n    Should in future such a use-case really appear, we have check_raw_mode() that\n    will make sure we implement support for it first.\n\n    Signed-off-by: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Acked-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 8683d7e6de79b0099bb98c7f5ea9590b2f49b200\nAuthor: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\nDate:   Wed Apr 13 00:10:50 2016 +0200\n\n    bpf, verifier: add bpf_call_arg_meta for passing meta data\n\n    Currently, when the verifier checks calls in check_call() function, we\n    call check_func_arg() for all 5 arguments e.g. to make sure expected types\n    are correct. In some cases, we collect meta data (here: map pointer) to\n    perform additional checks such as checking stack boundary on key/value\n    sizes for subsequent arguments. As we\u0027re going to extend the meta data,\n    add a generic struct bpf_call_arg_meta that we can use for passing into\n    check_func_arg().\n\n    Signed-off-by: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Acked-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 22249ef3025ba459847346d1b2bae3012caeee25\nAuthor: Alexei Starovoitov \u003cast@fb.com\u003e\nDate:   Tue Apr 12 10:26:19 2016 -0700\n\n    bpf/verifier: reject invalid LD_ABS | BPF_DW instruction\n\n    verifier must check for reserved size bits in instruction opcode and\n    reject BPF_LD | BPF_ABS | BPF_DW and BPF_LD | BPF_IND | BPF_DW instructions,\n    otherwise interpreter will WARN_RATELIMIT on them during execution.\n\n    Fixes: ddd872bc3098 (\"bpf: verifier: add checks for BPF_ABS | BPF_IND instructions\")\n    Signed-off-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Acked-by: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit ffecfe4243f1874eaa403286799be4a705c5c063\nAuthor: Alexei Starovoitov \u003cast@fb.com\u003e\nDate:   Wed Apr 6 19:39:21 2016 -0700\n\n    bpf: simplify verifier register state assignments\n\n    verifier is using the following structure to track the state of registers:\n    struct reg_state {\n        enum bpf_reg_type type;\n        union {\n            int imm;\n            struct bpf_map *map_ptr;\n        };\n    };\n    and later on in states_equal() does memcmp(\u0026old-\u003eregs[i], \u0026cur-\u003eregs[i],..)\n    to find equivalent states.\n    Throughout the code of verifier there are assignements to \u0027imm\u0027 and \u0027map_ptr\u0027\n    fields and it\u0027s not obvious that most of the assignments into \u0027imm\u0027 don\u0027t\n    need to clear extra 4 bytes (like mark_reg_unknown_value() does) to make sure\n    that memcmp doesn\u0027t go over junk left from \u0027map_ptr\u0027 assignment.\n\n    Simplify the code by converting \u0027int\u0027 into \u0027long\u0027\n\n    Suggested-by: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Signed-off-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Acked-by: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 37d179bc68b8cf62833ee391ac4da9d363a76845\nAuthor: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\nDate:   Tue Apr 5 22:33:17 2016 +0200\n\n    bpf, verifier: further improve search pruning\n\n    The verifier needs to go through every path of the program in\n    order to check that it terminates safely, which can be quite a\n    lot of instructions that need to be processed f.e. in cases with\n    more branchy programs. With search pruning from f1bca824dabb (\"bpf:\n    add search pruning optimization to verifier\") the search space can\n    already be reduced significantly when the verifier detects that\n    a previously walked path with same register and stack contents\n    terminated already (see verifier\u0027s states_equal()), so the search\n    can skip walking those states.\n\n    When working with larger programs of \u003e ~2000 (out of max 4096)\n    insns, we found that the current limit of 32k instructions is easily\n    hit. For example, a case we ran into is that the search space cannot\n    be pruned due to branches at the beginning of the program that make\n    use of certain stack space slots (STACK_MISC), which are never used\n    in the remaining program (STACK_INVALID). Therefore, the verifier\n    needs to walk paths for the slots in STACK_INVALID state, but also\n    all remaining paths with a stack structure, where the slots are in\n    STACK_MISC, which can nearly double the search space needed. After\n    various experiments, we find that a limit of 64k processed insns is\n    a more reasonable choice when dealing with larger programs in practice.\n    This still allows to reject extreme crafted cases that can have a\n    much higher complexity (f.e. \u003e ~300k) within the 4096 insns limit\n    due to search pruning not being able to take effect.\n\n    Furthermore, we found that a lot of states can be pruned after a\n    call instruction, f.e. we were able to reduce the search state by\n    ~35% in some cases with this heuristic, trade-off is to keep a bit\n    more states in env-\u003eexplored_states. Usually, call instructions\n    have a number of preceding register assignments and/or stack stores,\n    where search pruning has a better chance to suceed in states_equal()\n    test. The current code marks the branch targets with STATE_LIST_MARK\n    in case of conditional jumps, and the next (t + 1) instruction in\n    case of unconditional jump so that f.e. a backjump will walk it. We\n    also did experiments with using t + insns[t].off + 1 as a marker in\n    the unconditionally jump case instead of t + 1 with the rationale\n    that these two branches of execution that converge after the label\n    might have more potential of pruning. We found that it was a bit\n    better, but not necessarily significantly better than the current\n    state, perhaps also due to clang not generating back jumps often.\n    Hence, we left that as is for now.\n\n    Signed-off-by: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Acked-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 94a64d7f40731dc185f583fbe4f2a09568ae320c\nAuthor: Alexei Starovoitov \u003cast@fb.com\u003e\nDate:   Wed Apr 6 18:43:28 2016 -0700\n\n    bpf: sanitize bpf tracepoint access\n\n    during bpf program loading remember the last byte of ctx access\n    and at the time of attaching the program to tracepoint check that\n    the program doesn\u0027t access bytes beyond defined in tracepoint fields\n\n    This also disallows access to __dynamic_array fields, but can be\n    relaxed in the future.\n\n    Signed-off-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 69530735cc0048a79e0ea28f2f5ab3d0b6ad3580\nAuthor: Alexei Starovoitov \u003cast@fb.com\u003e\nDate:   Wed Apr 6 18:43:27 2016 -0700\n\n    bpf: support bpf_get_stackid() and bpf_perf_event_output() in tracepoint programs\n\n    needs two wrapper functions to fetch \u0027struct pt_regs *\u0027 to convert\n    tracepoint bpf context into kprobe bpf context to reuse existing\n    helper functions\n\n    Signed-off-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit db2a5763f32413dbc9148c5c4dc5ffbc0b967d78\nAuthor: Alexei Starovoitov \u003cast@fb.com\u003e\nDate:   Wed Apr 6 18:43:26 2016 -0700\n\n    bpf: register BPF_PROG_TYPE_TRACEPOINT program type\n\n    register tracepoint bpf program type and let it call the same set\n    of helper functions as BPF_PROG_TYPE_KPROBE\n\n    Signed-off-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 03365d848f0a3753b001bbb72c1e45b0f83534f8\nAuthor: Alexei Starovoitov \u003cast@fb.com\u003e\nDate:   Wed Apr 6 18:43:25 2016 -0700\n\n    perf, bpf: allow bpf programs attach to tracepoints\n\n    introduce BPF_PROG_TYPE_TRACEPOINT program type and allow it to be attached\n    to the perf tracepoint handler, which will copy the arguments into\n    the per-cpu buffer and pass it to the bpf program as its first argument.\n    The layout of the fields can be discovered by doing\n    \u0027cat /sys/kernel/debug/tracing/events/sched/sched_switch/format\u0027\n    prior to the compilation of the program with exception that first 8 bytes\n    are reserved and not accessible to the program. This area is used to store\n    the pointer to \u0027struct pt_regs\u0027 which some of the bpf helpers will use:\n    +---------+\n    | 8 bytes | hidden \u0027struct pt_regs *\u0027 (inaccessible to bpf program)\n    +---------+\n    | N bytes | static tracepoint fields defined in tracepoint/format (bpf readonly)\n    +---------+\n    | dynamic | __dynamic_array bytes of tracepoint (inaccessible to bpf yet)\n    +---------+\n\n    Not that all of the fields are already dumped to user space via perf ring buffer\n    and broken application access it directly without consulting tracepoint/format.\n    Same rule applies here: static tracepoint fields should only be accessed\n    in a format defined in tracepoint/format. The order of fields and\n    field sizes are not an ABI.\n\n    Signed-off-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Acked-by: Peter Zijlstra (Intel) \u003cpeterz@infradead.org\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 986f3aa294b51f80059c67b9bac4c80cbc0789eb\nAuthor: Alexei Starovoitov \u003cast@fb.com\u003e\nDate:   Wed Apr 6 18:43:24 2016 -0700\n\n    perf: split perf_trace_buf_prepare into alloc and update parts\n\n    split allows to move expensive update of \u0027struct trace_entry\u0027 to later phase.\n    Repurpose unused 1st argument of perf_tp_event() to indicate event type.\n\n    While splitting use temp variable \u0027rctx\u0027 instead of \u0027*rctx\u0027 to avoid\n    unnecessary loads done by the compiler due to -fno-strict-aliasing\n\n    Signed-off-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Acked-by: Peter Zijlstra (Intel) \u003cpeterz@infradead.org\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 1870873f44e69836a5677f302b85f0cc830b4a84\nAuthor: Alexei Starovoitov \u003cast@fb.com\u003e\nDate:   Wed Apr 6 18:43:23 2016 -0700\n\n    perf: remove unused __addr variable\n\n    now all calls to perf_trace_buf_submit() pass 0 as 4th\n    argument which will be repurposed in the next patch which will\n    change the meaning of 1st arg of perf_tp_event() to event_type\n\n    Signed-off-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Acked-by: Peter Zijlstra (Intel) \u003cpeterz@infradead.org\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit e7f515b54cb3de36385de8b42c2690531931be2f\nAuthor: Al Viro \u003cviro@zeniv.linux.org.uk\u003e\nDate:   Fri Mar 25 12:06:51 2016 -0400\n\n    bpf: reject invalid names right in -\u003elookup()\n\n    ... and other methods won\u0027t see them at all\n\n    Signed-off-by: Al Viro \u003cviro@zeniv.linux.org.uk\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 989de5ab3bd4b747a39a979e052203e3980593c4\nAuthor: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\nDate:   Fri Mar 25 00:30:25 2016 +0100\n\n    bpf: add missing map_flags to bpf_map_show_fdinfo\n\n    Add map_flags attribute to bpf_map_show_fdinfo(), so that tools like\n    tc can check for them when loading objects from a pinned entry, e.g.\n    if user intent wrt allocation (BPF_F_NO_PREALLOC) is different to the\n    pinned object, it can bail out. Follow-up to 6c9059817432 (\"bpf:\n    pre-allocate hash map elements\"), so that tc can still support this\n    with v4.6.\n\n    Signed-off-by: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Acked-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 6b10fb1009db25dd0d3c5dd7ecfd422591599413\nAuthor: Alexei Starovoitov \u003cast@fb.com\u003e\nDate:   Wed Mar 9 20:02:33 2016 -0800\n\n    bpf: avoid copying junk bytes in bpf_get_current_comm()\n\n    Lots of places in the kernel use memcpy(buf, comm, TASK_COMM_LEN); but\n    the result is typically passed to print(\"%s\", buf) and extra bytes\n    after zero don\u0027t cause any harm.\n    In bpf the result of bpf_get_current_comm() is used as the part of\n    map key and was causing spurious hash map mismatches.\n    Use strlcpy() to guarantee zero-terminated string.\n    bpf verifier checks that output buffer is zero-initialized,\n    so even for short task names the output buffer don\u0027t have junk bytes.\n    Note it\u0027s not a security concern, since kprobe+bpf is root only.\n\n    Fixes: ffeedafbf023 (\"bpf: introduce current-\u003epid, tgid, uid, gid, comm accessors\")\n    Reported-by: Tobias Waldekranz \u003ctobias@waldekranz.com\u003e\n    Signed-off-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit b1d5a61adf9e790e2d255158bb19706367de9329\nAuthor: Alexei Starovoitov \u003cast@fb.com\u003e\nDate:   Wed Mar 9 18:56:49 2016 -0800\n\n    bpf: bpf_stackmap_copy depends on CONFIG_PERF_EVENTS\n\n    0-day bot reported build error:\n    kernel/built-in.o: In function `map_lookup_elem\u0027:\n    \u003e\u003e kernel/bpf/.tmp_syscall.o:(.text+0x329b3c): undefined reference to `bpf_stackmap_copy\u0027\n    when CONFIG_BPF_SYSCALL is set and CONFIG_PERF_EVENTS is not.\n    Add weak definition to resolve it.\n    This code path in map_lookup_elem() is never taken\n    when CONFIG_PERF_EVENTS is not set.\n\n    Fixes: 557c0c6e7df8 (\"bpf: convert stackmap to pre-allocation\")\n    Reported-by: Fengguang Wu \u003cfengguang.wu@intel.com\u003e\n    Signed-off-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit d7f8276b4034c326e5770c41b4b81461b02e6e22\nAuthor: Alexei Starovoitov \u003cast@fb.com\u003e\nDate:   Mon Mar 7 21:57:17 2016 -0800\n\n    bpf: convert stackmap to pre-allocation\n\n    It was observed that calling bpf_get_stackid() from a kprobe inside\n    slub or from spin_unlock causes similar deadlock as with hashmap,\n    therefore convert stackmap to use pre-allocated memory.\n\n    The call_rcu is no longer feasible mechanism, since delayed freeing\n    causes bpf_get_stackid() to fail unpredictably when number of actual\n    stacks is significantly less than user requested max_entries.\n    Since elements are no longer freed into slub, we can push elements into\n    freelist immediately and let them be recycled.\n    However the very unlikley race between user space map_lookup() and\n    program-side recycling is possible:\n         cpu0                          cpu1\n         ----                          ----\n    user does lookup(stackidX)\n    starts copying ips into buffer\n                                       delete(stackidX)\n                                       calls bpf_get_stackid()\n    \t\t\t\t   which recyles the element and\n                                       overwrites with new stack trace\n\n    To avoid user space seeing a partial stack trace consisting of two\n    merged stack traces, do bucket \u003d xchg(, NULL); copy; xchg(,bucket);\n    to preserve consistent stack trace delivery to user space.\n    Now we can move memset(,0) of left-over element value from critical\n    path of bpf_get_stackid() into slow-path of user space lookup.\n    Also disallow lookup() from bpf program, since it\u0027s useless and\n    program shouldn\u0027t be messing with collected stack trace.\n\n    Note that similar race between user space lookup and kernel side updates\n    is also present in hashmap, but it\u0027s not a new race. bpf programs were\n    always allowed to modify hash and array map elements while user space\n    is copying them.\n\n    Fixes: d5a3b1f69186 (\"bpf: introduce BPF_MAP_TYPE_STACK_TRACE\")\n    Signed-off-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit c8e47381bd3af50fd4cde463003f44e9cf278a9b\nAuthor: Alexei Starovoitov \u003cast@fb.com\u003e\nDate:   Mon Mar 7 21:57:16 2016 -0800\n\n    bpf: check for reserved flag bits in array and stack maps\n\n    Suggested-by: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Signed-off-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit b5e87bad280370ef1096df8461836a634346a2c2\nAuthor: Alexei Starovoitov \u003cast@fb.com\u003e\nDate:   Mon Mar 7 21:57:15 2016 -0800\n\n    bpf: pre-allocate hash map elements\n\n    If kprobe is placed on spin_unlock then calling kmalloc/kfree from\n    bpf programs is not safe, since the following dead lock is possible:\n    kfree-\u003espin_lock(kmem_cache_node-\u003elock)...spin_unlock-\u003ekprobe-\u003e\n    bpf_prog-\u003emap_update-\u003ekmalloc-\u003espin_lock(of the same kmem_cache_node-\u003elock)\n    and deadlocks.\n\n    The following solutions were considered and some implemented, but\n    eventually discarded\n    - kmem_cache_create for every map\n    - add recursion check to slow-path of slub\n    - use reserved memory in bpf_map_update for in_irq or in preempt_disabled\n    - kmalloc via irq_work\n\n    At the end pre-allocation of all map elements turned out to be the simplest\n    solution and since the user is charged upfront for all the memory, such\n    pre-allocation doesn\u0027t affect the user space visible behavior.\n\n    Since it\u0027s impossible to tell whether kprobe is triggered in a safe\n    location from kmalloc point of view, use pre-allocation by default\n    and introduce new BPF_F_NO_PREALLOC flag.\n\n    While testing of per-cpu hash maps it was discovered\n    that alloc_percpu(GFP_ATOMIC) has odd corner cases and often\n    fails to allocate memory even when 90% of it is free.\n    The pre-allocation of per-cpu hash elements solves this problem as well.\n\n    Turned out that bpf_map_update() quickly followed by\n    bpf_map_lookup()+bpf_map_delete() is very common pattern used\n    in many of iovisor/bcc/tools, so there is additional benefit of\n    pre-allocation, since such use cases are must faster.\n\n    Since all hash map elements are now pre-allocated we can remove\n    atomic increment of htab-\u003ecount and save few more cycles.\n\n    Also add bpf_map_precharge_memlock() to check rlimit_memlock early to avoid\n    large malloc/free done by users who don\u0027t have sufficient limits.\n\n    Pre-allocation is done with vmalloc and alloc/free is done\n    via percpu_freelist. Here are performance numbers for different\n    pre-allocation algorithms that were implemented, but discarded\n    in favor of percpu_freelist:\n\n    1 cpu:\n    pcpu_ida\t2.1M\n    pcpu_ida nolock\t2.3M\n    bt\t\t2.4M\n    kmalloc\t\t1.8M\n    hlist+spinlock\t2.3M\n    pcpu_freelist\t2.6M\n\n    4 cpu:\n    pcpu_ida\t1.5M\n    pcpu_ida nolock\t1.8M\n    bt w/smp_align\t1.7M\n    bt no/smp_align\t1.1M\n    kmalloc\t\t0.7M\n    hlist+spinlock\t0.2M\n    pcpu_freelist\t2.0M\n\n    8 cpu:\n    pcpu_ida\t0.7M\n    bt w/smp_align\t0.8M\n    kmalloc\t\t0.4M\n    pcpu_freelist\t1.5M\n\n    32 cpu:\n    kmalloc\t\t0.13M\n    pcpu_freelist\t0.49M\n\n    pcpu_ida nolock is a modified percpu_ida algorithm without\n    percpu_ida_cpu locks and without cross-cpu tag stealing.\n    It\u0027s faster than existing percpu_ida, but not as fast as pcpu_freelist.\n\n    bt is a variant of block/blk-mq-tag.c simlified and customized\n    for bpf use case. bt w/smp_align is using cache line for every \u0027long\u0027\n    (similar to blk-mq-tag). bt no/smp_align allocates \u0027long\u0027\n    bitmasks continuously to save memory. It\u0027s comparable to percpu_ida\n    and in some cases faster, but slower than percpu_freelist\n\n    hlist+spinlock is the simplest free list with single spinlock.\n    As expeceted it has very bad scaling in SMP.\n\n    kmalloc is existing implementation which is still available via\n    BPF_F_NO_PREALLOC flag. It\u0027s significantly slower in single cpu and\n    in 8 cpu setup it\u0027s 3 times slower than pre-allocation with pcpu_freelist,\n    but saves memory, so in cases where map-\u003emax_entries can be large\n    and number of map update/delete per second is low, it may make\n    sense to use it.\n\n    Signed-off-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 8ab0075ed6148a1c03aa0cbdb46d87d80dffcffc\nAuthor: Alexei Starovoitov \u003cast@fb.com\u003e\nDate:   Mon Mar 7 21:57:14 2016 -0800\n\n    bpf: introduce percpu_freelist\n\n    Introduce simple percpu_freelist to keep single list of elements\n    spread across per-cpu singly linked lists.\n\n    /* push element into the list */\n    void pcpu_freelist_push(struct pcpu_freelist *, struct pcpu_freelist_node *);\n\n    /* pop element from the list */\n    struct pcpu_freelist_node *pcpu_freelist_pop(struct pcpu_freelist *);\n\n    The object is pushed to the current cpu list.\n    Pop first trying to get the object from the current cpu list,\n    if it\u0027s empty goes to the neigbour cpu list.\n\n    For bpf program usage pattern the collision rate is very low,\n    since programs push and pop the objects typically on the same cpu.\n\n    Signed-off-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit a003f77a9e901023841778cc4388294fb537d650\nAuthor: Alexei Starovoitov \u003cast@fb.com\u003e\nDate:   Mon Mar 7 21:57:13 2016 -0800\n\n    bpf: prevent kprobe+bpf deadlocks\n\n    if kprobe is placed within update or delete hash map helpers\n    that hold bucket spin lock and triggered bpf program is trying to\n    grab the spinlock for the same bucket on the same cpu, it will\n    deadlock.\n    Fix it by extending existing recursion prevention mechanism.\n\n    Note, map_lookup and other tracing helpers don\u0027t have this problem,\n    since they don\u0027t hold any locks and don\u0027t modify global data.\n    bpf_trace_printk has its own recursive check and ok as well.\n\n    Signed-off-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Acked-by: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit d9cd772dca5eb3a4e03fa868aa2d6add480d817a\nAuthor: Josh Poimboeuf \u003cjpoimboe@redhat.com\u003e\nDate:   Sun Feb 28 22:22:37 2016 -0600\n\n    bpf: Mark __bpf_prog_run() stack frame as non-standard\n\n    objtool reports the following false positive warnings:\n\n      kernel/bpf/core.o: warning: objtool: __bpf_prog_run()+0x5c: sibling call from callable instruction with changed frame pointer\n      kernel/bpf/core.o: warning: objtool: __bpf_prog_run()+0x60: function has unreachable instruction\n      kernel/bpf/core.o: warning: objtool: __bpf_prog_run()+0x64: function has unreachable instruction\n      [...]\n\n    It\u0027s confused by the following dynamic jump instruction in\n    __bpf_prog_run()::\n\n      jmp     *(%r12,%rax,8)\n\n    which corresponds to the following line in the C code:\n\n      goto *jumptable[insn-\u003ecode];\n\n    There\u0027s no way for objtool to deterministically find all possible\n    branch targets for a dynamic jump, so it can\u0027t verify this code.\n\n    In this case the jumps all stay within the function, and there\u0027s nothing\n    unusual going on related to the stack, so we can whitelist the function.\n\n    Signed-off-by: Josh Poimboeuf \u003cjpoimboe@redhat.com\u003e\n    Acked-by: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Acked-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Cc: Andrew Morton \u003cakpm@linux-foundation.org\u003e\n    Cc: Andy Lutomirski \u003cluto@kernel.org\u003e\n    Cc: Arnaldo Carvalho de Melo \u003cacme@kernel.org\u003e\n    Cc: Bernd Petrovitsch \u003cbernd@petrovitsch.priv.at\u003e\n    Cc: Borislav Petkov \u003cbp@alien8.de\u003e\n    Cc: Chris J Arges \u003cchris.j.arges@canonical.com\u003e\n    Cc: Jiri Slaby \u003cjslaby@suse.cz\u003e\n    Cc: Linus Torvalds \u003ctorvalds@linux-foundation.org\u003e\n    Cc: Michal Marek \u003cmmarek@suse.cz\u003e\n    Cc: Namhyung Kim \u003cnamhyung@gmail.com\u003e\n    Cc: Pedro Alves \u003cpalves@redhat.com\u003e\n    Cc: Peter Zijlstra \u003cpeterz@infradead.org\u003e\n    Cc: Thomas Gleixner \u003ctglx@linutronix.de\u003e\n    Cc: live-patching@vger.kernel.org\n    Cc: netdev@vger.kernel.org\n    Link: http://lkml.kernel.org/r/b90e6bf3fdbfb5c4cc1b164b965502e53cf48935.1456719558.git.jpoimboe@redhat.com\n    Signed-off-by: Ingo Molnar \u003cmingo@kernel.org\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 33df8c4ab343871610c65197264e8cc09ef04612\nAuthor: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\nDate:   Fri Feb 19 23:05:22 2016 +0100\n\n    bpf: add new arg_type that allows for 0 sized stack buffer\n\n    Currently, when we pass a buffer from the eBPF stack into a helper\n    function, the function proto indicates argument types as ARG_PTR_TO_STACK\n    and ARG_CONST_STACK_SIZE pair. If R\u003cX\u003e contains the former, then R\u003cX+1\u003e\n    must be of the latter type. Then, verifier checks whether the buffer\n    points into eBPF stack, is initialized, etc. The verifier also guarantees\n    that the constant value passed in R\u003cX+1\u003e is greater than 0, so helper\n    functions don\u0027t need to test for it and can always assume a non-NULL\n    initialized buffer as well as non-0 buffer size.\n\n    This patch adds a new argument types ARG_CONST_STACK_SIZE_OR_ZERO that\n    allows to also pass NULL as R\u003cX\u003e and 0 as R\u003cX+1\u003e into the helper function.\n    Such helper functions, of course, need to be able to handle these cases\n    internally then. Verifier guarantees that either R\u003cX\u003e \u003d\u003d NULL \u0026\u0026 R\u003cX+1\u003e \u003d\u003d 0\n    or R\u003cX\u003e !\u003d NULL \u0026\u0026 R\u003cX+1\u003e !\u003d 0 (like the case of ARG_CONST_STACK_SIZE), any\n    other combinations are not possible to load.\n\n    I went through various options of extending the verifier, and introducing\n    the type ARG_CONST_STACK_SIZE_OR_ZERO seems to have most minimal changes\n    needed to the verifier.\n\n    Signed-off-by: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Acked-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 43e79743a6b6d486b3977fd0c281be531096ab5a\nAuthor: Alexei Starovoitov \u003cast@fb.com\u003e\nDate:   Wed Feb 17 19:58:58 2016 -0800\n\n    bpf: introduce BPF_MAP_TYPE_STACK_TRACE\n\n    add new map type to store stack traces and corresponding helper\n    bpf_get_stackid(ctx, map, flags) - walk user or kernel stack and return id\n    @ctx: struct pt_regs*\n    @map: pointer to stack_trace map\n    @flags: bits 0-7 - numer of stack frames to skip\n            bit 8 - collect user stack instead of kernel\n            bit 9 - compare stacks by hash only\n            bit 10 - if two different stacks hash into the same stackid\n                     discard old\n            other bits - reserved\n    Return: \u003e\u003d 0 stackid on success or negative error\n\n    stackid is a 32-bit integer handle that can be further combined with\n    other data (including other stackid) and used as a key into maps.\n\n    Userspace will access stackmap using standard lookup/delete syscall commands to\n    retrieve full stack trace for given stackid.\n\n    Signed-off-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 4c16860477822114cf19d85c45c61db73e5d88e3\nAuthor: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\nDate:   Mon Jan 11 01:16:39 2016 +0100\n\n    bpf: support ipv6 for bpf_skb_{set,get}_tunnel_key\n\n    After IPv6 support has recently been added to metadata dst and related\n    encaps, add support for populating/reading it from an eBPF program.\n\n    Commit d3aa45ce6b (\"bpf: add helpers to access tunnel metadata\") started\n    with initial IPv4-only support back then (due to IPv6 metadata support\n    not being available yet).\n\n    To stay compatible with older programs, we need to test for the passed\n    structure size. Also TOS and TTL support from the ip_tunnel_info key has\n    been added. Tested with vxlan devs in collect meta data mode with IPv4,\n    IPv6 and in compat mode over different network namespaces.\n\n    Signed-off-by: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Acked-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 3e5d2da2428d9a2ebc15119db5de5be9b59f857e\nAuthor: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\nDate:   Mon Jan 11 01:16:38 2016 +0100\n\n    bpf: export helper function flags and reject invalid ones\n\n    Export flags used by eBPF helper functions through UAPI, so they can be\n    used by programs (instead of them redefining all flags each time or just\n    using the hard-coded values). It also gives a better overview what flags\n    are used where and we can further get rid of the extra macros defined in\n    filter.c. Moreover, reject invalid flags.\n\n    Signed-off-by: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Acked-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 702bf27a7cf50f7496f476041e6c404f9ebf49c5\nAuthor: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\nDate:   Thu Jan 7 15:50:23 2016 +0100\n\n    bpf: add skb_postpush_rcsum and fix dev_forward_skb occasions\n\n    Add a small helper skb_postpush_rcsum() and fix up redirect locations\n    that need CHECKSUM_COMPLETE fixups on ingress. dev_forward_skb() expects\n    a proper csum that covers also Ethernet header, f.e. since 2c26d34bbcc0\n    (\"net/core: Handle csum for CHECKSUM_COMPLETE VXLAN forwarding\"), we\n    also do skb_postpull_rcsum() after pulling Ethernet header off via\n    eth_type_trans().\n\n    When using eBPF in a netns setup f.e. with vxlan in collect metadata mode,\n    I can trigger the following csum issue with an IPv6 setup:\n\n      [  505.144065] dummy1: hw csum failure\n      [...]\n      [  505.144108] Call Trace:\n      [  505.144112]  \u003cIRQ\u003e  [\u003cffffffff81372f08\u003e] dump_stack+0x44/0x5c\n      [  505.144134]  [\u003cffffffff81607cea\u003e] netdev_rx_csum_fault+0x3a/0x40\n      [  505.144142]  [\u003cffffffff815fee3f\u003e] __skb_checksum_complete+0xcf/0xe0\n      [  505.144149]  [\u003cffffffff816f0902\u003e] nf_ip6_checksum+0xb2/0x120\n      [  505.144161]  [\u003cffffffffa08c0e0e\u003e] icmpv6_error+0x17e/0x328 [nf_conntrack_ipv6]\n      [  505.144170]  [\u003cffffffffa0898eca\u003e] ? ip6t_do_table+0x2fa/0x645 [ip6_tables]\n      [  505.144177]  [\u003cffffffffa08c0725\u003e] ? ipv6_get_l4proto+0x65/0xd0 [nf_conntrack_ipv6]\n      [  505.144189]  [\u003cffffffffa06c9a12\u003e] nf_conntrack_in+0xc2/0x5a0 [nf_conntrack]\n      [  505.144196]  [\u003cffffffffa08c039c\u003e] ipv6_conntrack_in+0x1c/0x20 [nf_conntrack_ipv6]\n      [  505.144204]  [\u003cffffffff8164385d\u003e] nf_iterate+0x5d/0x70\n      [  505.144210]  [\u003cffffffff816438d6\u003e] nf_hook_slow+0x66/0xc0\n      [  505.144218]  [\u003cffffffff816bd302\u003e] ipv6_rcv+0x3f2/0x4f0\n      [  505.144225]  [\u003cffffffff816bca40\u003e] ? ip6_make_skb+0x1b0/0x1b0\n      [  505.144232]  [\u003cffffffff8160b77b\u003e] __netif_receive_skb_core+0x36b/0x9a0\n      [  505.144239]  [\u003cffffffff8160bdc8\u003e] ? __netif_receive_skb+0x18/0x60\n      [  505.144245]  [\u003cffffffff8160bdc8\u003e] __netif_receive_skb+0x18/0x60\n      [  505.144252]  [\u003cffffffff8160ccff\u003e] process_backlog+0x9f/0x140\n      [  505.144259]  [\u003cffffffff8160c4a5\u003e] net_rx_action+0x145/0x320\n      [...]\n\n    What happens is that on ingress, we push Ethernet header back in, either\n    from cls_bpf or right before skb_do_redirect(), but without updating csum.\n    The \"hw csum failure\" can be fixed by using the new skb_postpush_rcsum()\n    helper for the dev_forward_skb() case to correct the csum diff again.\n\n    Thanks to Hannes Frederic Sowa for the csum_partial() idea!\n\n    Fixes: 3896d655f4d4 (\"bpf: introduce bpf_clone_redirect() helper\")\n    Fixes: 27b29f63058d (\"bpf: add bpf_redirect() helper\")\n    Signed-off-by: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Acked-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 1c1049b97190ebeb4ff711f76db6f89c6aa68b92\nAuthor: Craig Gallek \u003ckraig@google.com\u003e\nDate:   Mon Jan 4 17:41:47 2016 -0500\n\n    soreuseport: setsockopt SO_ATTACH_REUSEPORT_[CE]BPF\n\n    Expose socket options for setting a classic or extended BPF program\n    for use when selecting sockets in an SO_REUSEPORT group.  These options\n    can be used on the first socket to belong to a group before bind or\n    on any socket in the group after bind.\n\n    This change includes refactoring of the existing sk_filter code to\n    allow reuse of the existing BPF filter validation checks.\n\n    Signed-off-by: Craig Gallek \u003ckraig@google.com\u003e\n    Acked-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n    Change-Id: If4bd3258a57eeca4622a135d9cdf64ecc12f8a27\n\ncommit 3a46d6c7aeeac4486b65f682a1b6e7e97e7740a6\nAuthor: Craig Gallek \u003ckraig@google.com\u003e\nDate:   Mon Jan 4 17:41:46 2016 -0500\n\n    soreuseport: fast reuseport UDP socket selection\n\n    Include a struct sock_reuseport instance when a UDP socket binds to\n    a specific address for the first time with the reuseport flag set.\n    When selecting a socket for an incoming UDP packet, use the information\n    available in sock_reuseport if present.\n\n    This required adding an additional field to the UDP source address\n    equality function to differentiate between exact and wildcard matches.\n    The original use case allowed wildcard matches when checking for\n    existing port uses during bind.  The new use case of adding a socket\n    to a reuseport group requires exact address matching.\n\n    Performance test (using a machine with 2 CPU sockets and a total of\n    48 cores):  Create reuseport groups of varying size.  Use one socket\n    from this group per user thread (pinning each thread to a different\n    core) calling recvmmsg in a tight loop.  Record number of messages\n    received per second while saturating a 10G link.\n      10 sockets: 18% increase (~2.8M -\u003e 3.3M pkts/s)\n      20 sockets: 14% increase (~2.9M -\u003e 3.3M pkts/s)\n      40 sockets: 13% increase (~3.0M -\u003e 3.4M pkts/s)\n\n    This work is based off a similar implementation written by\n    Ying Cai \u003cycai@google.com\u003e for implementing policy-based reuseport\n    selection.\n\n    Signed-off-by: Craig Gallek \u003ckraig@google.com\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 93cf2d5634671797a50a13a428db67e34d249183\nAuthor: Craig Gallek \u003ckraig@google.com\u003e\nDate:   Mon Jan 4 17:41:45 2016 -0500\n\n    soreuseport: define reuseport groups\n\n    struct sock_reuseport is an optional shared structure referenced by each\n    socket belonging to a reuseport group.  When a socket is bound to an\n    address/port not yet in use and the reuseport flag has been set, the\n    structure will be allocated and attached to the newly bound socket.\n    When subsequent calls to bind are made for the same address/port, the\n    shared structure will be updated to include the new socket and the\n    newly bound socket will reference the group structure.\n\n    Usually, when an incoming packet was destined for a reuseport group,\n    all sockets in the same group needed to be considered before a\n    dispatching decision was made.  With this structure, an appropriate\n    socket can be found after looking up just one socket in the group.\n\n    This shared structure will also allow for more complicated decisions to\n    be made when selecting a socket (eg a BPF filter).\n\n    This work is based off a similar implementation written by\n    Ying Cai \u003cycai@google.com\u003e for implementing policy-based reuseport\n    selection.\n\n    Signed-off-by: Craig Gallek \u003ckraig@google.com\u003e\n    Acked-by: Eric Dumazet \u003cedumazet@google.com\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit e85ab7e8c3669f6f35a5d056aa9157485d7e17d0\nAuthor: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\nDate:   Thu Dec 17 23:51:55 2015 +0100\n\n    bpf: fix misleading comment in bpf_convert_filter\n\n    Comment says \"User BPF\u0027s register A is mapped to our BPF register 6\",\n    which is actually wrong as the mapping is on register 0. This can\n    already be inferred from the code itself. So just remove it before\n    someone makes assumptions based on that. Only code tells truth. ;)\n\n    Signed-off-by: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Acked-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 081e8e95f12d42e1c37c20d9efd254fb48a56f4b\nAuthor: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\nDate:   Thu Dec 17 23:51:53 2015 +0100\n\n    bpf: add bpf_skb_load_bytes helper\n\n    When hacking tc programs with eBPF, one of the issues that come up\n    from time to time is to load addresses from headers. In eBPF as in\n    classic BPF, we have BPF_LD | BPF_ABS | BPF_{B,H,W} instructions that\n    extract a byte, half-word or word out of the skb data though helpers\n    such as bpf_load_pointer() (interpreter case).\n\n    F.e. extracting a whole IPv6 address could possibly look like ...\n\n      union v6addr {\n        struct {\n          __u32 p1;\n          __u32 p2;\n          __u32 p3;\n          __u32 p4;\n        };\n        __u8 addr[16];\n      };\n\n      [...]\n\n      a.p1 \u003d htonl(load_word(skb, off));\n      a.p2 \u003d htonl(load_word(skb, off +  4));\n      a.p3 \u003d htonl(load_word(skb, off +  8));\n      a.p4 \u003d htonl(load_word(skb, off + 12));\n\n      [...]\n\n      /* access to a.addr[...] */\n\n    This work adds a complementary helper bpf_skb_load_bytes() (we also\n    have bpf_skb_store_bytes()) as an alternative where the same call\n    would look like from an eBPF program:\n\n      ret \u003d bpf_skb_load_bytes(skb, off, addr, sizeof(addr));\n\n    Same verifier restrictions apply as in ffeedafbf023 (\"bpf: introduce\n    current-\u003epid, tgid, uid, gid, comm accessors\") case, where stack memory\n    access needs to be statically verified and thus guaranteed to be\n    initialized in first use (otherwise verifier cannot tell whether a\n    subsequent access to it is valid or not as it\u0027s runtime dependent).\n\n    Signed-off-by: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Acked-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit fa6dc8f675002d5d5d1d77be5fc430542b972bdb\nAuthor: Sasha Levin \u003csasha.levin@oracle.com\u003e\nDate:   Fri Feb 19 13:53:10 2016 -0500\n\n    bpf: grab rcu read lock for bpf_percpu_hash_update\n\n    bpf_percpu_hash_update() expects rcu lock to be held and warns if it\u0027s not,\n    which pointed out a missing rcu read lock.\n\n    Fixes: 15a07b338 (\"bpf: add lookup/update support for per-cpu hash and array maps\")\n    Signed-off-by: Sasha Levin \u003csasha.levin@oracle.com\u003e\n    Acked-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Acked-by: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 3aa373d0a667e624f95159af7fd40e8920cb2e92\nAuthor: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\nDate:   Wed Feb 10 16:47:11 2016 +0100\n\n    bpf: fix branch offset adjustment on backjumps after patching ctx expansion\n\n    When ctx access is used, the kernel often needs to expand/rewrite\n    instructions, so after that patching, branch offsets have to be\n    adjusted for both forward and backward jumps in the new eBPF program,\n    but for backward jumps it fails to account the delta. Meaning, for\n    example, if the expansion happens exactly on the insn that sits at\n    the jump target, it doesn\u0027t fix up the back jump offset.\n\n    Analysis on what the check in adjust_branches() is currently doing:\n\n      /* adjust offset of jmps if necessary */\n      if (i \u003c pos \u0026\u0026 i + insn-\u003eoff + 1 \u003e pos)\n        insn-\u003eoff +\u003d delta;\n      else if (i \u003e pos \u0026\u0026 i + insn-\u003eoff + 1 \u003c pos)\n        insn-\u003eoff -\u003d delta;\n\n    First condition (forward jumps):\n\n      Before:                         After:\n\n      insns[0]                        insns[0]\n      insns[1] \u003c--- i/insn            insns[1] \u003c--- i/insn\n      insns[2] \u003c--- pos               insns[P] \u003c--- pos\n      insns[3]                        insns[P]  `------| delta\n      insns[4] \u003c--- target_X          insns[P]   `-----|\n      insns[5]                        insns[3]\n                                      insns[4] \u003c--- target_X\n                                      insns[5]\n\n    First case is if we cross pos-boundary and the jump instruction was\n    before pos. This is handeled correctly. I.e. if i \u003d\u003d pos, then this\n    would mean our jump that we currently check was the patchlet itself\n    that we just injected. Since such patchlets are self-contained and\n    have no awareness of any insns before or after the patched one, the\n    delta is correctly not adjusted. Also, for the second condition in\n    case of i + insn-\u003eoff + 1 \u003d\u003d pos, means we jump to that newly patched\n    instruction, so no offset adjustment are needed. That part is correct.\n\n    Second condition (backward jumps):\n\n      Before:                         After:\n\n      insns[0]                        insns[0]\n      insns[1] \u003c--- target_X          insns[1] \u003c--- target_X\n      insns[2] \u003c--- pos \u003c-- target_Y  insns[P] \u003c--- pos \u003c-- target_Y\n      insns[3]                        insns[P]  `------| delta\n      insns[4] \u003c--- i/insn            insns[P]   `-----|\n      insns[5]                        insns[3]\n                                      insns[4] \u003c--- i/insn\n                                      insns[5]\n\n    Second interesting case is where we cross pos-boundary and the jump\n    instruction was after pos. Backward jump with i \u003d\u003d pos would be\n    impossible and pose a bug somewhere in the patchlet, so the first\n    condition checking i \u003e pos is okay only by itself. However, i +\n    insn-\u003eoff + 1 \u003c pos does not always work as intended to trigger the\n    adjustment. It works when jump targets would be far off where the\n    delta wouldn\u0027t matter. But, for example, where the fixed insn-\u003eoff\n    before pointed to pos (target_Y), it now points to pos + delta, so\n    that additional room needs to be taken into account for the check.\n    This means that i) both tests here need to be adjusted into pos + delta,\n    and ii) for the second condition, the test needs to be \u003c\u003d as pos\n    itself can be a target in the backjump, too.\n\n    Fixes: 9bac3d6d548e (\"bpf: allow extended BPF programs access skb fields\")\n    Signed-off-by: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit ca132c344ca9dfdddf927c4cc9cd1790f1849000\nAuthor: Alexei Starovoitov \u003cast@fb.com\u003e\nDate:   Mon Feb 1 22:39:55 2016 -0800\n\n    bpf: add lookup/update support for per-cpu hash and array maps\n\n    The functions bpf_map_lookup_elem(map, key, value) and\n    bpf_map_update_elem(map, key, value, flags) need to get/set\n    values from all-cpus for per-cpu hash and array maps,\n    so that user space can aggregate/update them as necessary.\n\n    Example of single counter aggregation in user space:\n      unsigned int nr_cpus \u003d sysconf(_SC_NPROCESSORS_CONF);\n      long values[nr_cpus];\n      long value \u003d 0;\n\n      bpf_lookup_elem(fd, key, values);\n      for (i \u003d 0; i \u003c nr_cpus; i++)\n        value +\u003d values[i];\n\n    The user space must provide round_up(value_size, 8) * nr_cpus\n    array to get/set values, since kernel will use \u0027long\u0027 copy\n    of per-cpu values to try to copy good counters atomically.\n    It\u0027s a best-effort, since bpf programs and user space are racing\n    to access the same memory.\n\n    Signed-off-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit c795e49cf732a91cc4413809ade72b79be2bb632\nAuthor: Alexei Starovoitov \u003cast@fb.com\u003e\nDate:   Mon Feb 1 22:39:54 2016 -0800\n\n    bpf: introduce BPF_MAP_TYPE_PERCPU_ARRAY map\n\n    Primary use case is a histogram array of latency\n    where bpf program computes the latency of block requests or other\n    events and stores histogram of latency into array of 64 elements.\n    All cpus are constantly running, so normal increment is not accurate,\n    bpf_xadd causes cache ping-pong and this per-cpu approach allows\n    fastest collision-free counters.\n\n    Signed-off-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 04052cc31a398c54f25b6cf08eb0734a52942b9f\nAuthor: Alexei Starovoitov \u003cast@fb.com\u003e\nDate:   Mon Feb 1 22:39:53 2016 -0800\n\n    bpf: introduce BPF_MAP_TYPE_PERCPU_HASH map\n\n    Introduce BPF_MAP_TYPE_PERCPU_HASH map type which is used to do\n    accurate counters without need to use BPF_XADD instruction which turned\n    out to be too costly for high-performance network monitoring.\n    In the typical use case the \u0027key\u0027 is the flow tuple or other long\n    living object that sees a lot of events per second.\n\n    bpf_map_lookup_elem() returns per-cpu area.\n    Example:\n    struct {\n      u32 packets;\n      u32 bytes;\n    } * ptr \u003d bpf_map_lookup_elem(\u0026map, \u0026key);\n    /* ptr points to this_cpu area of the value, so the following\n     * increments will not collide with other cpus\n     */\n    ptr-\u003epackets ++;\n    ptr-\u003ebytes +\u003d skb-\u003elen;\n\n    bpf_update_elem() atomically creates a new element where all per-cpu\n    values are zero initialized and this_cpu value is populated with\n    given \u0027value\u0027.\n    Note that non-per-cpu hash map always allocates new element\n    and then deletes old after rcu grace period to maintain atomicity\n    of update. Per-cpu hash map updates element values in-place.\n\n    Signed-off-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 7b4e308a152a3754f5eb52eec2996b44801dde1c\nAuthor: Alexei Starovoitov \u003calexei.starovoitov@gmail.com\u003e\nDate:   Mon Jan 25 20:59:49 2016 -0800\n\n    perf/bpf: Convert perf_event_array to use struct file\n\n    Robustify refcounting.\n\n    Signed-off-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Signed-off-by: Peter Zijlstra (Intel) \u003cpeterz@infradead.org\u003e\n    Cc: Alexander Shishkin \u003calexander.shishkin@linux.intel.com\u003e\n    Cc: Arnaldo Carvalho de Melo \u003cacme@infradead.org\u003e\n    Cc: Arnaldo Carvalho de Melo \u003cacme@redhat.com\u003e\n    Cc: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Cc: David Ahern \u003cdsahern@gmail.com\u003e\n    Cc: Jiri Olsa \u003cjolsa@kernel.org\u003e\n    Cc: Jiri Olsa \u003cjolsa@redhat.com\u003e\n    Cc: Linus Torvalds \u003ctorvalds@linux-foundation.org\u003e\n    Cc: Namhyung Kim \u003cnamhyung@kernel.org\u003e\n    Cc: Peter Zijlstra \u003cpeterz@infradead.org\u003e\n    Cc: Stephane Eranian \u003ceranian@google.com\u003e\n    Cc: Thomas Gleixner \u003ctglx@linutronix.de\u003e\n    Cc: Vince Weaver \u003cvincent.weaver@maine.edu\u003e\n    Cc: Wang Nan \u003cwangnan0@huawei.com\u003e\n    Cc: vince@deater.net\n    Link: http://lkml.kernel.org/r/20160126045947.GA40151@ast-mbp.thefacebook.com\n    Signed-off-by: Ingo Molnar \u003cmingo@kernel.org\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 44d7aa703ab5024b0ddb9493f1a0aecbd2046213\nAuthor: Rabin Vincent \u003crabin@rab.in\u003e\nDate:   Tue Jan 12 20:17:08 2016 +0100\n\n    net: bpf: reject invalid shifts\n\n    On ARM64, a BUG() is triggered in the eBPF JIT if a filter with a\n    constant shift that can\u0027t be encoded in the immediate field of the\n    UBFM/SBFM instructions is passed to the JIT.  Since these shifts\n    amounts, which are negative or \u003e\u003d regsize, are invalid, reject them in\n    the eBPF verifier and the classic BPF filter checker, for all\n    architectures.\n\n    Signed-off-by: Rabin Vincent \u003crabin@rab.in\u003e\n    Acked-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Acked-by: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 5d82c86f0a0fd5da784ee9c1b16e404ae8323918\nAuthor: tom.leiming@gmail.com \u003ctom.leiming@gmail.com\u003e\nDate:   Tue Dec 29 22:40:27 2015 +0800\n\n    bpf: hash: use per-bucket spinlock\n\n    Both htab_map_update_elem() and htab_map_delete_elem() can be\n    called from eBPF program, and they may be in kernel hot path,\n    so it isn\u0027t efficient to use a per-hashtable lock in this two\n    helpers.\n\n    The per-hashtable spinlock is used for protecting bucket\u0027s\n    hlist, and per-bucket lock is just enough. This patch converts\n    the per-hashtable lock into per-bucket spinlock, so that\n    contention can be decreased a lot.\n\n    Signed-off-by: Ming Lei \u003ctom.leiming@gmail.com\u003e\n    Acked-by: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 65a9b4946bbf2e1e330a9aa6c5e51c7badfb36e9\nAuthor: tom.leiming@gmail.com \u003ctom.leiming@gmail.com\u003e\nDate:   Tue Dec 29 22:40:26 2015 +0800\n\n    bpf: hash: move select_bucket() out of htab\u0027s spinlock\n\n    The spinlock is just used for protecting the per-bucket\n    hlist, so it isn\u0027t needed for selecting bucket.\n\n    Acked-by: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Signed-off-by: Ming Lei \u003ctom.leiming@gmail.com\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit c8e06496fe4a361c06396c925593f02441dca246\nAuthor: tom.leiming@gmail.com \u003ctom.leiming@gmail.com\u003e\nDate:   Tue Dec 29 22:40:25 2015 +0800\n\n    bpf: hash: use atomic count\n\n    Preparing for removing global per-hashtable lock, so\n    the counter need to be defined as aotmic_t first.\n\n    Acked-by: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Signed-off-by: Ming Lei \u003ctom.leiming@gmail.com\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 5da59e4e0bd28d85f8b94cd7e511265dd749b418\nAuthor: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\nDate:   Thu Dec 17 23:51:54 2015 +0100\n\n    bpf: move clearing of A/X into classic to eBPF migration prologue\n\n    Back in the days where eBPF (or back then \"internal BPF\" ;-\u003e) was not\n    exposed to user space, and only the classic BPF programs internally\n    translated into eBPF programs, we missed the fact that for classic BPF\n    A and X needed to be cleared. It was fixed back then via 83d5b7ef99c9\n    (\"net: filter: initialize A and X registers\"), and thus classic BPF\n    specifics were added to the eBPF interpreter core to work around it.\n\n    This added some confusion for JIT developers later on that take the\n    eBPF interpreter code as an example for deriving their JIT. F.e. in\n    f75298f5c3fe (\"s390/bpf: clear correct BPF accumulator register\"), at\n    least X could leak stack memory. Furthermore, since this is only needed\n    for classic BPF translations and not for eBPF (verifier takes care\n    that read access to regs cannot be done uninitialized), more complexity\n    is added to JITs as they need to determine whether they deal with\n    migrations or native eBPF where they can just omit clearing A/X in\n    their prologue and thus reduce image size a bit, see f.e. cde66c2d88da\n    (\"s390/bpf: Only clear A and X for converted BPF programs\"). In other\n    cases (x86, arm64), A and X is being cleared in the prologue also for\n    eBPF case, which is unnecessary.\n\n    Lets move this into the BPF migration in bpf_convert_filter() where it\n    actually belongs as long as the number of eBPF JITs are still few. It\n    can thus be done generically; allowing us to remove the quirk from\n    __bpf_prog_run() and to slightly reduce JIT image size in case of eBPF,\n    while reducing code duplication on this matter in current(/future) eBPF\n    JITs.\n\n    Signed-off-by: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Acked-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Reviewed-by: Michael Holzheu \u003cholzheu@linux.vnet.ibm.com\u003e\n    Tested-by: Michael Holzheu \u003cholzheu@linux.vnet.ibm.com\u003e\n    Cc: Zi Shen Lim \u003czlim.lnx@gmail.com\u003e\n    Cc: Yang Shi \u003cyang.shi@linaro.org\u003e\n    Acked-by: Yang Shi \u003cyang.shi@linaro.org\u003e\n    Acked-by: Zi Shen Lim \u003czlim.lnx@gmail.com\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 514b5beddd33546aa9bb320c60f6ed317a78a6bc\nAuthor: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\nDate:   Thu Dec 10 22:33:49 2015 +0100\n\n    bpf, inode: allow for rename and link ops\n\n    Add support for renaming and hard links to the fs. Most of this can be\n    implemented by using simple library operations under the same constraints\n    that we don\u0027t use a reserved name like elsewhere. Linking can be useful\n    to share/manage things like maps across subsystem users. It works within\n    the file system boundary, but is not allowed for directories.\n\n    Symbolic links are explicitly not implemented here, as it can be better\n    done already by doing bind mounts inside bpf fs to set up shared directories\n    f.e. useful when using volumes in docker containers that map a private\n    working directory into /sys/fs/bpf/ which contains itself a bind mounted\n    path from the host\u0027s /sys/fs/bpf/ mount that is shared among multiple\n    containers. For single maps instead of whole directory, hard links can\n    be easily used to do the same.\n\n    Signed-off-by: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Acked-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 9c45cc504c3d4e42abdbc78414525f892a3b5708\nAuthor: Alexei Starovoitov \u003cast@kernel.org\u003e\nDate:   Sun Nov 29 16:59:35 2015 -0800\n\n    bpf: fix allocation warnings in bpf maps and integer overflow\n\n    For large map-\u003evalue_size the user space can trigger memory allocation warnings like:\n    WARNING: CPU: 2 PID: 11122 at mm/page_alloc.c:2989\n    __alloc_pages_nodemask+0x695/0x14e0()\n    Call Trace:\n     [\u003c     inline     \u003e] __dump_stack lib/dump_stack.c:15\n     [\u003cffffffff82743b56\u003e] dump_stack+0x68/0x92 lib/dump_stack.c:50\n     [\u003cffffffff81244ec9\u003e] warn_slowpath_common+0xd9/0x140 kernel/panic.c:460\n     [\u003cffffffff812450f9\u003e] warn_slowpath_null+0x29/0x30 kernel/panic.c:493\n     [\u003c     inline     \u003e] __alloc_pages_slowpath mm/page_alloc.c:2989\n     [\u003cffffffff81554e95\u003e] __alloc_pages_nodemask+0x695/0x14e0 mm/page_alloc.c:3235\n     [\u003cffffffff816188fe\u003e] alloc_pages_current+0xee/0x340 mm/mempolicy.c:2055\n     [\u003c     inline     \u003e] alloc_pages include/linux/gfp.h:451\n     [\u003cffffffff81550706\u003e] alloc_kmem_pages+0x16/0xf0 mm/page_alloc.c:3414\n     [\u003cffffffff815a1c89\u003e] kmalloc_order+0x19/0x60 mm/slab_common.c:1007\n     [\u003cffffffff815a1cef\u003e] kmalloc_order_trace+0x1f/0xa0 mm/slab_common.c:1018\n     [\u003c     inline     \u003e] kmalloc_large include/linux/slab.h:390\n     [\u003cffffffff81627784\u003e] __kmalloc+0x234/0x250 mm/slub.c:3525\n     [\u003c     inline     \u003e] kmalloc include/linux/slab.h:463\n     [\u003c     inline     \u003e] map_update_elem kernel/bpf/syscall.c:288\n     [\u003c     inline     \u003e] SYSC_bpf kernel/bpf/syscall.c:744\n\n    To avoid never succeeding kmalloc with order \u003e\u003d MAX_ORDER check that\n    elem-\u003evalue_size and computed elem_size are within limits for both hash and\n    array type maps.\n    Also add __GFP_NOWARN to kmalloc(value_size | elem_size) to avoid OOM warnings.\n    Note kmalloc(key_size) is highly unlikely to trigger OOM, since key_size \u003c\u003d 512,\n    so keep those kmalloc-s as-is.\n\n    Large value_size can cause integer overflows in elem_size and map.pages\n    formulas, so check for that as well.\n\n    Fixes: aaac3ba95e4c (\"bpf: charge user for creation of BPF maps and programs\")\n    Reported-by: Dmitry Vyukov \u003cdvyukov@google.com\u003e\n    Signed-off-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 93e43b4d010f09485fddc91885055f90081883eb\nAuthor: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\nDate:   Mon Nov 30 13:02:56 2015 +0100\n\n    bpf, array: fix heap out-of-bounds access when updating elements\n\n    During own review but also reported by Dmitry\u0027s syzkaller [1] it has been\n    noticed that we trigger a heap out-of-bounds access on eBPF array maps\n    when updating elements. This happens with each map whose map-\u003evalue_size\n    (specified during map creation time) is not multiple of 8 bytes.\n\n    In array_map_alloc(), elem_size is round_up(attr-\u003evalue_size, 8) and\n    used to align array map slots for faster access. However, in function\n    array_map_update_elem(), we update the element as ...\n\n    memcpy(array-\u003evalue + array-\u003eelem_size * index, value, array-\u003eelem_size);\n\n    ... where we access \u0027value\u0027 out-of-bounds, since it was allocated from\n    map_update_elem() from syscall side as kmalloc(map-\u003evalue_size, GFP_USER)\n    and later on copied through copy_from_user(value, uvalue, map-\u003evalue_size).\n    Thus, up to 7 bytes, we can access out-of-bounds.\n\n    Same could happen from within an eBPF program, where in worst case we\n    access beyond an eBPF program\u0027s designated stack.\n\n    Since 1be7f75d1668 (\"bpf: enable non-root eBPF programs\") didn\u0027t hit an\n    official release yet, it only affects priviledged users.\n\n    In case of array_map_lookup_elem(), the verifier prevents eBPF programs\n    from accessing beyond map-\u003evalue_size through check_map_access(). Also\n    from syscall side map_lookup_elem() only copies map-\u003evalue_size back to\n    user, so nothing could leak.\n\n      [1] http://github.com/google/syzkaller\n\n    Fixes: 28fbcfa08d8e (\"bpf: add array type of eBPF maps\")\n    Reported-by: Dmitry Vyukov \u003cdvyukov@google.com\u003e\n    Signed-off-by: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Acked-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 5fd3bbeeb5b76baf6fb8878377a9d25cf51b8b52\nAuthor: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\nDate:   Tue Nov 24 21:28:15 2015 +0100\n\n    bpf: fix clearing on persistent program array maps\n\n    Currently, when having map file descriptors pointing to program arrays,\n    there\u0027s still the issue that we unconditionally flush program array\n    contents via bpf_fd_array_map_clear() in bpf_map_release(). This happens\n    when such a file descriptor is released and is independent of the map\u0027s\n    refcount.\n\n    Having this flush independent of the refcount is for a reason: there\n    can be arbitrary complex dependency chains among tail calls, also circular\n    ones (direct or indirect, nesting limit determined during runtime), and\n    we need to make sure that the map drops all references to eBPF programs\n    it holds, so that the map\u0027s refcount can eventually drop to zero and\n    initiate its freeing. Btw, a walk of the whole dependency graph would\n    not be possible for various reasons, one being complexity and another\n    one inconsistency, i.e. new programs can be added to parts of the graph\n    at any time, so there\u0027s no guaranteed consistent state for the time of\n    such a walk.\n\n    Now, the program array pinning itself works, but the issue is that each\n    derived file descriptor on close would nevertheless call unconditionally\n    into bpf_fd_array_map_clear(). Instead, keep track of users and postpone\n    this flush until the last reference to a user is dropped. As this only\n    concerns a subset of references (f.e. a prog array could hold a program\n    that itself has reference on the prog array holding it, etc), we need to\n    track them separately.\n\n    Short analysis on the refcounting: on map creation time usercnt will be\n    one, so there\u0027s no change in behaviour for bpf_map_release(), if unpinned.\n    If we already fail in map_create(), we are immediately freed, and no\n    file descriptor has been made public yet. In bpf_obj_pin_user(), we need\n    to probe for a possible map in bpf_fd_probe_obj() already with a usercnt\n    reference, so before we drop the reference on the fd with fdput().\n    Therefore, if actual pinning fails, we need to drop that reference again\n    in bpf_any_put(), otherwise we keep holding it. When last reference\n    drops on the inode, the bpf_any_put() in bpf_evict_inode() will take\n    care of dropping the usercnt again. In the bpf_obj_get_user() case, the\n    bpf_any_get() will grab a reference on the usercnt, still at a time when\n    we have the reference on the path. Should we later on fail to grab a new\n    file descriptor, bpf_any_put() will drop it, otherwise we hold it until\n    bpf_map_release() time.\n\n    Joint work with Alexei.\n\n    Fixes: b2197755b263 (\"bpf: add support for persistent maps/progs\")\n    Signed-off-by: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Signed-off-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 4b3f084abf47da6140d5a96d5083ec4ef7be35d1\nAuthor: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\nDate:   Thu Nov 19 11:56:22 2015 +0100\n\n    bpf: add show_fdinfo handler for maps\n\n    Add a handler for show_fdinfo() to be used by the anon-inodes\n    backend for eBPF maps, and dump the map specification there. Not\n    only useful for admins, but also it provides a minimal way to\n    compare specs from ELF vs pinned object.\n\n    Signed-off-by: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Acked-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Acked-by: Hannes Frederic Sowa \u003channes@stressinduktion.org\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit b993170dfb6532765cb1f94774ffbd5d83af94b6\nAuthor: Anay Wadhera \u003cawadhera@berkeley.edu\u003e\nDate:   Thu May 20 22:00:28 2021 -0700\n\n    Revert \"bpf: fix clearing on persistent program array maps\"\n\n    This reverts commit c9da161c6517ba12154059d3b965c2cbaf16f90f.\n\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 6735bc67368515c04c2e5338e6d5738c7b10ef39\nAuthor: Anay Wadhera \u003cawadhera@berkeley.edu\u003e\nDate:   Thu May 20 22:00:23 2021 -0700\n\n    Revert \"bpf, array: fix heap out-of-bounds access when updating elements\"\n\n    This reverts commit fbca9d2d35c6ef1b323fae75cc9545005ba25097.\n\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 609e0835b15f33be0a50025be8acd2280a6068be\nAuthor: Anay Wadhera \u003cawadhera@berkeley.edu\u003e\nDate:   Thu May 20 22:00:16 2021 -0700\n\n    Revert \"bpf: fix allocation warnings in bpf maps and integer overflow\"\n\n    This reverts commit 01b3f52157ff5a47d6d8d796f396a4b34a53c61d.\n\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 928bd46a86c51f023a10c2f54d6cd2eaf06ca212\nAuthor: Anay Wadhera \u003cawadhera@berkeley.edu\u003e\nDate:   Thu May 20 22:00:11 2021 -0700\n\n    Revert \"net: bpf: reject invalid shifts\"\n\n    This reverts commit 35987ff2eaa05d70154c5bd28ebb2b70a7d8368b.\n\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 4cf55928c5a61a7045df3e999223874a3aff80d2\nAuthor: Anay Wadhera \u003cawadhera@berkeley.edu\u003e\nDate:   Thu May 20 22:00:06 2021 -0700\n\n    Revert \"bpf: fix branch offset adjustment on backjumps after patching ctx expansion\"\n\n    This reverts commit a34f2f9f2034f7984f9529002c6fffe9cb63189d.\n\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit b92f8a7ee0fc76df59b3af27d067cf5d35a0044e\nAuthor: Anay Wadhera \u003cawadhera@berkeley.edu\u003e\nDate:   Thu May 20 22:00:00 2021 -0700\n\n    Revert \"bpf: avoid copying junk bytes in bpf_get_current_comm()\"\n\n    This reverts commit e8e43232627082328fa4016fab1960360360f167.\n\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 05182c79d6472324ff4f9bdf076ca9d80c406f67\nAuthor: Anay Wadhera \u003cawadhera@berkeley.edu\u003e\nDate:   Thu May 20 21:59:51 2021 -0700\n\n    Revert \"bpf/verifier: reject invalid LD_ABS | BPF_DW instruction\"\n\n    This reverts commit 8427d5547d0b63beb70d3858127942f828400ad2.\n\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 88d2d86f1e5d53edeb29a3d06b3b94b6320dc700\nAuthor: Anay Wadhera \u003cawadhera@berkeley.edu\u003e\nDate:   Thu May 20 21:58:52 2021 -0700\n\n    Revert \"bpf: fix double-fdput in replace_map_fd_with_map_ptr()\"\n\n    This reverts commit 608d2c3c7a046c222cae2e857cf648a9f89e772b.\n\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 0acc470ed6c52a66f6175d73f9fe54d8648f554e\nAuthor: Anay Wadhera \u003cawadhera@berkeley.edu\u003e\nDate:   Thu May 20 21:58:41 2021 -0700\n\n    Revert \"bpf: fix refcnt overflow\"\n\n    This reverts commit 3899251bdb9c2b31fc73d4cc132f52d3710101de.\n\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 82e4ad585e90f5827a80c7e1fea7d38aee371200\nAuthor: Anay Wadhera \u003cawadhera@berkeley.edu\u003e\nDate:   Thu May 20 21:58:40 2021 -0700\n\n    Revert \"bpf: fix check_map_func_compatibility logic\"\n\n    This reverts commit bb10156f572f06f3b6cadd378e5a0ab3ed8da991.\n\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 437814d65d6be82a89c7d19c2a2d1dda7109455c\nAuthor: Anay Wadhera \u003cawadhera@berkeley.edu\u003e\nDate:   Thu May 20 21:58:12 2021 -0700\n\n    Revert \"bpf: Use mount_nodev not mount_ns to mount the bpf filesystem\"\n\n    This reverts commit 5b7ea922e1754107f77d146011612f2e42600cc1.\n\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit a2d50cfa44dab239102312a80b5f69c1206a0188\nAuthor: Anay Wadhera \u003cawadhera@berkeley.edu\u003e\nDate:   Thu May 20 21:58:07 2021 -0700\n\n    Revert \"bpf, inode: disallow userns mounts\"\n\n    This reverts commit bfe951d547bf15bf1192abd20773e6603dacadf1.\n\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 069f5b2aa8f89a5556fc531958476bdd156aec61\nAuthor: Anay Wadhera \u003cawadhera@berkeley.edu\u003e\nDate:   Thu May 20 21:58:02 2021 -0700\n\n    Revert \"bpf: prevent leaking pointer via xadd on unpriviledged\"\n\n    This reverts commit 1a4f13e0a99a85c455ff2f6dc117f6f049c039fa.\n\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 654d00c0b8287f809c30507b3a08a2e394855e00\nAuthor: Anay Wadhera \u003cawadhera@berkeley.edu\u003e\nDate:   Thu May 20 21:57:55 2021 -0700\n\n    Revert \"bpf/verifier: reject BPF_ALU64|BPF_END\"\n\n    This reverts commit 2ec54b21dd7b25df0f070f1d67db2ea18987e69e.\n\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 34467768210f021f0aec7dabff5b9e0c6b182376\nAuthor: Anay Wadhera \u003cawadhera@berkeley.edu\u003e\nDate:   Thu May 20 21:57:48 2021 -0700\n\n    Revert \"bpf: don\u0027t let ldimm64 leak map addresses on unprivileged\"\n\n    This reverts commit 49630dd2e10a3b2fee0cec19feb63f08453b876f.\n\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 37ce3f871f002c3a340e2ff6560116ea55049666\nAuthor: Anay Wadhera \u003cawadhera@berkeley.edu\u003e\nDate:   Thu May 20 21:56:46 2021 -0700\n\n    Revert \"bpf: add bpf_patch_insn_single helper\"\n\n    This reverts commit 087a92287dbae61b4ee1e76d7c20c81710109422.\n\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 46c1980e529e4870fcfc95b5406b09bb279c4bcf\nAuthor: Anay Wadhera \u003cawadhera@berkeley.edu\u003e\nDate:   Thu May 20 21:56:46 2021 -0700\n\n    Revert \"bpf: don\u0027t (ab)use instructions to store state\"\n\n    This reverts commit 0748b80e432584502d1559b1a51b7df58f5e2fce.\n\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit e4d6ebc747526631101b7e1112686b2189f4104e\nAuthor: Anay Wadhera \u003cawadhera@berkeley.edu\u003e\nDate:   Thu May 20 21:56:45 2021 -0700\n\n    Revert \"bpf: move fixup_bpf_calls() function\"\n\n    This reverts commit 14c7c55f452740549d561e583714b700cd88883e.\n\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit d7a869af8d0aa0d5436e2c8859a58ca746e6548a\nAuthor: Anay Wadhera \u003cawadhera@berkeley.edu\u003e\nDate:   Thu May 20 21:56:45 2021 -0700\n\n    Revert \"bpf: refactor fixup_bpf_calls()\"\n\n    This reverts commit 19614eee0644a59a8ea2509a6fbc0e771644a4f2.\n\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit a99ed9ef738b420625a3c9e8441043f672488bf2\nAuthor: Anay Wadhera \u003cawadhera@berkeley.edu\u003e\nDate:   Thu May 20 21:56:44 2021 -0700\n\n    Revert \"bpf: adjust insn_aux_data when patching insns\"\n\n    This reverts commit 648064515d0d91d10d255ab1e3afa3ecffc2943a.\n\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 4f5a8642bc49946f822dd2162bbaeadd64fbf5ee\nAuthor: Anay Wadhera \u003cawadhera@berkeley.edu\u003e\nDate:   Thu May 20 21:56:43 2021 -0700\n\n    Revert \"bpf: prevent out-of-bounds speculation\"\n\n    This reverts commit 9a7fad4c0e215fb1c256fee27c45f9f8bc4364c5.\n\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 8efde2b259578f829b95a1ce90b0a1261624eebc\nAuthor: Anay Wadhera \u003cawadhera@berkeley.edu\u003e\nDate:   Thu May 20 21:56:42 2021 -0700\n\n    Revert \"bpf, array: fix overflow in max_entries and undefined behavior in index_mask\"\n\n    This reverts commit 095b0ba360ff9a86c592c1293602d42a9297e047.\n\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 96569a03ca9dfbbcbdbfc93fce7b80bba423f2b9\nAuthor: Anay Wadhera \u003cawadhera@berkeley.edu\u003e\nDate:   Thu May 20 21:56:29 2021 -0700\n\n    Revert \"bpf: fix branch pruning logic\"\n\n    This reverts commit 1367d854b97493bfb1f3d24cf89ba60cb7f059ea.\n\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit da51eafe038dbda178b1435530faf663c7003db9\nAuthor: Anay Wadhera \u003cawadhera@berkeley.edu\u003e\nDate:   Thu May 20 21:56:22 2021 -0700\n\n    Revert \"bpf: fix bpf_tail_call() x64 JIT\"\n\n    This reverts commit 361fb0481247bea4da3eb122e685c8b72ef7c8a9.\n\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 857717ca02275bedfdfce1574bc2fd3c44486135\nAuthor: Anay Wadhera \u003cawadhera@berkeley.edu\u003e\nDate:   Thu May 20 21:55:49 2021 -0700\n\n    Revert \"bpf: introduce BPF_JIT_ALWAYS_ON config\"\n\n    This reverts commit 28c486744e6de4d882a1d853aa63d99fcba4b7a6.\n\n    Change-Id: Iffebc366a5c2cc47b16e7a09438b018485facb95\n\ncommit df9200d0917b01ae70a79b34ba483574718d5638\nAuthor: Anay Wadhera \u003cawadhera@berkeley.edu\u003e\nDate:   Thu May 20 21:55:29 2021 -0700\n\n    Revert \"bpf: arsh is not supported in 32 bit alu thus reject it\"\n\n    This reverts commit 7dcda40e52ff0712a2d7d5353c1722cb1f994330.\n\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 056335ef4d9ae9aa22757f2883848fe34241c59d\nAuthor: Anay Wadhera \u003cawadhera@berkeley.edu\u003e\nDate:   Thu May 20 21:55:29 2021 -0700\n\n    Revert \"bpf: avoid false sharing of map refcount with max_entries\"\n\n    This reverts commit 96d9b2338bed553c37f759127d8d18c857449ceb.\n\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 1c238b1a8e9a451be0764abc5e1b7cd0efea6afd\nAuthor: Anay Wadhera \u003cawadhera@berkeley.edu\u003e\nDate:   Thu May 20 21:55:28 2021 -0700\n\n    Revert \"bpf: fix divides by zero\"\n\n    This reverts commit b72ba2a0d82447538c7c977ccb3f2b31b19b7767.\n\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 447a58b830cce8c59d997de357f91101a60d5000\nAuthor: Anay Wadhera \u003cawadhera@berkeley.edu\u003e\nDate:   Thu May 20 21:55:27 2021 -0700\n\n    Revert \"bpf: fix 32-bit divide by zero\"\n\n    This reverts commit 02662601a231f8721930168ce71d84bcfb8d9a96.\n\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit ca6816795eebb210375fabb5566f22be9379b301\nAuthor: Anay Wadhera \u003cawadhera@berkeley.edu\u003e\nDate:   Thu May 20 21:55:26 2021 -0700\n\n    Revert \"bpf: reject stores into ctx via st and xadd\"\n\n    This reverts commit faa74a862a9442233bff39a496013a74775fb660.\n\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 0377b7697f26620e44a05d6cf49d5f9023ebd99c\nAuthor: Anay Wadhera \u003cawadhera@berkeley.edu\u003e\nDate:   Thu May 20 21:50:08 2021 -0700\n\n    Revert \"bpf: fix incorrect sign extension in check_alu_op()\"\n\n    This reverts commit a6132276ab5dcc38b3299082efeb25b948263adb.\n\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 0a4be322061ab87402dd9c59f45629f350a72be6\nAuthor: Anay Wadhera \u003cawadhera@berkeley.edu\u003e\nDate:   Thu May 20 21:50:03 2021 -0700\n\n    Revert \"bpf: skip unnecessary capability check\"\n\n    This reverts commit c9ea2f8af67399904fe9c72ab5192a0c0ae7f2bf.\n\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 5e042bd8e7f4485b185085f3a65a07db01d6d2b8\nAuthor: Anay Wadhera \u003cawadhera@berkeley.edu\u003e\nDate:   Thu May 20 21:49:58 2021 -0700\n\n    Revert \"bpf: map_get_next_key to return first key on NULL\"\n\n    This reverts commit ea7c24c78551c8b3e6a7e9824e5ad8ba6224f5fe.\n\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 007816bf4ab563a10e55bf85abd225beab0f60ab\nAuthor: Anay Wadhera \u003cawadhera@berkeley.edu\u003e\nDate:   Thu May 20 21:49:02 2021 -0700\n\n    Revert \"bpf: fix references to free_bpf_prog_info() in comments\"\n\n    This reverts commit b23dab51e987787e358397b24831505668625b8a.\n\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit fcbcf061c0db8a369c2ec1428d45a50a42e22a4c\nAuthor: Anay Wadhera \u003cawadhera@berkeley.edu\u003e\nDate:   Thu May 20 21:48:56 2021 -0700\n\n    Revert \"bpf: generally move prog destruction to RCU deferral\"\n\n    This reverts commit e25dc63aa366fd0f61d1d9ba67b66f5d75fc4372.\n\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit c4c8e5a91e53eb7b9a0b194d3722dae8473eeaec\nAuthor: Anay Wadhera \u003cawadhera@berkeley.edu\u003e\nDate:   Thu May 20 21:48:49 2021 -0700\n\n    Revert \"bpf: support 8-byte metafield access\"\n\n    This reverts commit 3c4bb079e16e222324c68d7594b1ab6f699edfca.\n\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit b3a8f286bb9f9cf251221624f1987c0014bd927e\nAuthor: Anay Wadhera \u003cawadhera@berkeley.edu\u003e\nDate:   Thu May 20 21:48:48 2021 -0700\n\n    Revert \"bpf/verifier: Add spi variable to check_stack_write()\"\n\n    This reverts commit 168cb9b7b2839e861278f9fde03820aba32c4ee0.\n\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 813c93faf78e24fd78229f9fc1bcaf3ef07886d6\nAuthor: Anay Wadhera \u003cawadhera@berkeley.edu\u003e\nDate:   Thu May 20 21:48:47 2021 -0700\n\n    Revert \"bpf/verifier: Pass instruction index to check_mem_access() and check_xadd()\"\n\n    This reverts commit 451624d47005aace4e314b488cb70ba3ee5dcce8.\n\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 0c67c1301cc50d5a7a69c6751ba0027347a55539\nAuthor: Anay Wadhera \u003cawadhera@berkeley.edu\u003e\nDate:   Thu May 20 21:48:46 2021 -0700\n\n    Revert \"bpf: Prevent memory disambiguation attack\"\n\n    This reverts commit 1c74bd22e846b162ea6401e8d43172e0e7256ccf.\n\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 3d948c96e6babcdbf1d3bdab3dbd0935c5a9f7f3\nAuthor: Anay Wadhera \u003cawadhera@berkeley.edu\u003e\nDate:   Thu May 20 21:48:24 2021 -0700\n\n    Revert \"bpf: silence warning messages in core\"\n\n    This reverts commit 7dd2dc652435c0abb9f05ff9ef0b378fcf743f10.\n\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 33a28fdec2c5763953393b0a4d5b397f4595f655\nAuthor: Georg Veichtlbauer \u003cgeorg@vware.at\u003e\nDate:   Fri Feb 11 20:43:28 2022 +0100\n\n    Revert \"cgroup: replace __DEVEL__sane_behavior with cgroup2 fs type\"\n\n    This reverts commit cd5367ae02a488450b714a5d5f47afb014ea6541.\n\n    Change-Id: I3d43cceedfde5bf08053ba3ed806992cfc2d8523\n\ncommit 679ee5a4e643ab09c9cedef80cf5e4f61990d120\nAuthor: Stephen Smalley \u003csds@tycho.nsa.gov\u003e\nDate:   Fri Apr 8 13:52:00 2016 -0400\n\n    selinux: distinguish non-init user namespace capability checks\n\n    Distinguish capability checks against a target associated\n    with the init user namespace versus capability checks against\n    a target associated with a non-init user namespace by defining\n    and using separate security classes for the latter.\n\n    This is needed to support e.g. Chrome usage of user namespaces\n    for the Chrome sandbox without needing to allow Chrome to also\n    exercise capabilities on targets in the init user namespace.\n\n    Suggested-by: Dan Walsh \u003cdwalsh@redhat.com\u003e\n    Signed-off-by: Stephen Smalley \u003csds@tycho.nsa.gov\u003e\n    Signed-off-by: Paul Moore \u003cpaul@paul-moore.com\u003e\n    Change-Id: I6b56d3262a73dd8a410785a51e5048aab2c5e254\n\nChange-Id: Iab6c63ba26730279d2699057ea1747d52e2a20fa\n","web_links":[{"name":"GitHub","tooltip":"Open in GitWeb","url":"https://github.com/LineageOS/android_kernel_oneplus_msm8998/commit/838b49504e7c00f10a4cd9cf83d34036f4af3090"}],"resolve_conflicts_web_links":[{"name":"GitHub","tooltip":"Open in GitWeb","url":"https://github.com/LineageOS/android_kernel_oneplus_msm8998/commit/838b49504e7c00f10a4cd9cf83d34036f4af3090"}]},"branch":"refs/heads/lineage-19.0"},"230f727125d0cd1fc52a69ab72e29cbac151e74a":{"kind":"REWORK","_number":2,"created":"2022-02-12 06:17:58.000000000","uploader":{"_account_id":5911,"name":"Georg Veichtlbauer","email":"georg@vware.at","username":"veichtlbauer","avatars":[{"url":"https://www.gravatar.com/avatar/39db6f16bd92d063f8a1762ba2009d16.jpg?d\u003didenticon\u0026r\u003dpg\u0026s\u003d32","height":32},{"url":"https://www.gravatar.com/avatar/39db6f16bd92d063f8a1762ba2009d16.jpg?d\u003didenticon\u0026r\u003dpg\u0026s\u003d56","height":56},{"url":"https://www.gravatar.com/avatar/39db6f16bd92d063f8a1762ba2009d16.jpg?d\u003didenticon\u0026r\u003dpg\u0026s\u003d100","height":100},{"url":"https://www.gravatar.com/avatar/39db6f16bd92d063f8a1762ba2009d16.jpg?d\u003didenticon\u0026r\u003dpg\u0026s\u003d120","height":120}]},"ref":"refs/changes/55/323455/2","fetch":{"anonymous http":{"url":"https://github.com/LineageOS/android_kernel_oneplus_msm8998","ref":"refs/changes/55/323455/2","commands":{"Branch":"git fetch https://github.com/LineageOS/android_kernel_oneplus_msm8998 refs/changes/55/323455/2 \u0026\u0026 git checkout -b change-323455 FETCH_HEAD","Checkout":"git fetch https://github.com/LineageOS/android_kernel_oneplus_msm8998 refs/changes/55/323455/2 \u0026\u0026 git checkout FETCH_HEAD","Cherry Pick":"git fetch https://github.com/LineageOS/android_kernel_oneplus_msm8998 refs/changes/55/323455/2 \u0026\u0026 git cherry-pick FETCH_HEAD","Format Patch":"git fetch https://github.com/LineageOS/android_kernel_oneplus_msm8998 refs/changes/55/323455/2 \u0026\u0026 git format-patch -1 --stdout FETCH_HEAD","Pull":"git pull https://github.com/LineageOS/android_kernel_oneplus_msm8998 refs/changes/55/323455/2","Reset To":"git fetch https://github.com/LineageOS/android_kernel_oneplus_msm8998 refs/changes/55/323455/2 \u0026\u0026 git reset --hard FETCH_HEAD"}}},"commit":{"parents":[{"commit":"d3dee68c56d18f064801d121e6c734b4cb806b80","subject":"Merge branch \u0027google/android-4.4-p\u0027 into lineage-18.1","web_links":[{"name":"GitHub","tooltip":"Open in GitWeb","url":"https://github.com/LineageOS/android_kernel_oneplus_msm8998/commit/d3dee68c56d18f064801d121e6c734b4cb806b80"}]}],"author":{"name":"Georg Veichtlbauer","email":"georg@vware.at","date":"2022-02-12 06:15:07.000000000","tz":60},"committer":{"name":"Georg Veichtlbauer","email":"georg@vware.at","date":"2022-02-12 06:15:07.000000000","tz":60},"subject":"Backport BPF","message":"Backport BPF\n\ncommit ab36739971f70d5e28ddb19f4b59b4df2f061fb1\nAuthor: ivanmeler \u003ci_ivan@windowslive.com\u003e\nDate:   Tue Oct 26 17:25:51 2021 +0000\n\n    oneplus5: Remove wireguard\n\n    Change-Id: Ib9d794d2fd88c9d583a1c23e3f24313546abcffc\n\ncommit bcc6a8c96aaef404ec772e97f019680c4a489db1\nAuthor: ivanmeler \u003ci_ivan@windowslive.com\u003e\nDate:   Thu Oct 28 10:00:00 2021 +0000\n\n    oneplus5: Enable BPF\n\n    Change-Id: I3695f04c93f159ed7afce7a865aafd70a24818a3\n\ncommit 83db010a92623cc82f8af55204f243ffa437fdd0\nAuthor: Chatur27 \u003cjasonbright2709@gmail.com\u003e\nDate:   Sun Oct 10 19:36:47 2021 +0000\n\n    treewide: Fixup for BPF backport\n\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n    Signed-off-by: roynatech2544 \u003cwhiteshell2544@naver.com\u003e\n\ncommit 18aca682e9564eb2780f77cc3c8919e2b0a734ce\nAuthor: Arnd Bergmann \u003carnd@arndb.de\u003e\nDate:   Tue Sep 6 15:10:22 2016 +0200\n\n    perf, bpf: fix conditional call to bpf_overflow_handler\n\n    The newly added bpf_overflow_handler function is only built of both\n    CONFIG_EVENT_TRACING and CONFIG_BPF_SYSCALL are enabled, but the caller\n    only checks the latter:\n\n    kernel/events/core.c: In function \u0027perf_event_alloc\u0027:\n    kernel/events/core.c:9106:27: error: \u0027bpf_overflow_handler\u0027 undeclared (first use in this function)\n\n    This changes the caller so we also skip this call if CONFIG_EVENT_TRACING\n    is disabled entirely.\n\n    Signed-off-by: Arnd Bergmann \u003carnd@arndb.de\u003e\n    Fixes: aa6a5f3cb2b2 (\"perf, bpf: add perf events core support for BPF_PROG_TYPE_PERF_EVENT programs\")\n    Acked-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n\ncommit bb3bc0bb892b5c2f5f1a64b5311efa1ca9af8103\nAuthor: Maitreya29 \u003cMaitreyapatni30@gmail.com\u003e\nDate:   Mon Aug 30 11:35:04 2021 +0530\n\n    net: adapt bpf_xdp_copy\n\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit fef4fce8b3e985fc39813c0d5c802f8a83bcbfd0\nAuthor: Maitreya29 \u003cMaitreyapatni30@gmail.com\u003e\nDate:   Mon Aug 30 10:55:49 2021 +0530\n\n    {net, kernel}: Guard proc_dointvec_minmax_bpf_restricted and nuke void *priv from cpuset_fork\n\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit d4885fee112b361a3844294fe32726de718f7207\nAuthor: Maitreya29 \u003cMaitreyapatni30@gmail.com\u003e\nDate:   Sun Aug 29 21:53:14 2021 +0530\n\n    Revert \"net/compat: Add missing sock updates for SCM_RIGHTS\"\n\n    This reverts commit 34c2166235171162c55ccdc2f3f77b377da76d7c.\n\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 3b09a64b0598f3b6d296db7dfff3d8a0426afe2b\nAuthor: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\nDate:   Tue Dec 11 12:14:12 2018 +0100\n\n    bpf: fix bpf_jit_limit knob for PAGE_SIZE \u003e\u003d 64K\n\n    [ Upstream commit fdadd04931c2d7cd294dc5b2b342863f94be53a3 ]\n\n    Michael and Sandipan report:\n\n      Commit ede95a63b5 introduced a bpf_jit_limit tuneable to limit BPF\n      JIT allocations. At compile time it defaults to PAGE_SIZE * 40000,\n      and is adjusted again at init time if MODULES_VADDR is defined.\n\n      For ppc64 kernels, MODULES_VADDR isn\u0027t defined, so we\u0027re stuck with\n      the compile-time default at boot-time, which is 0x9c400000 when\n      using 64K page size. This overflows the signed 32-bit bpf_jit_limit\n      value:\n\n      root@ubuntu:/tmp# cat /proc/sys/net/core/bpf_jit_limit\n      -1673527296\n\n      and can cause various unexpected failures throughout the network\n      stack. In one case `strace dhclient eth0` reported:\n\n      setsockopt(5, SOL_SOCKET, SO_ATTACH_FILTER, {len\u003d11, filter\u003d0x105dd27f8},\n                 16) \u003d -1 ENOTSUPP (Unknown error 524)\n\n      and similar failures can be seen with tools like tcpdump. This doesn\u0027t\n      always reproduce however, and I\u0027m not sure why. The more consistent\n      failure I\u0027ve seen is an Ubuntu 18.04 KVM guest booted on a POWER9\n      host would time out on systemd/netplan configuring a virtio-net NIC\n      with no noticeable errors in the logs.\n\n    Given this and also given that in near future some architectures like\n    arm64 will have a custom area for BPF JIT image allocations we should\n    get rid of the BPF_JIT_LIMIT_DEFAULT fallback / default entirely. For\n    4.21, we have an overridable bpf_jit_alloc_exec(), bpf_jit_free_exec()\n    so therefore add another overridable bpf_jit_alloc_exec_limit() helper\n    function which returns the possible size of the memory area for deriving\n    the default heuristic in bpf_jit_charge_init().\n\n    Like bpf_jit_alloc_exec() and bpf_jit_free_exec(), the new\n    bpf_jit_alloc_exec_limit() assumes that module_alloc() is the default\n    JIT memory provider, and therefore in case archs implement their custom\n    module_alloc() we use MODULES_{END,_VADDR} for limits and otherwise for\n    vmalloc_exec() cases like on ppc64 we use VMALLOC_{END,_START}.\n\n    Additionally, for archs supporting large page sizes, we should change\n    the sysctl to be handled as long to not run into sysctl restrictions\n    in future.\n\n    Fixes: ede95a63b5e8 (\"bpf: add bpf_jit_limit knob to restrict unpriv allocations\")\n    Reported-by: Sandipan Das \u003csandipan@linux.ibm.com\u003e\n    Reported-by: Michael Roth \u003cmdroth@linux.vnet.ibm.com\u003e\n    Signed-off-by: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Tested-by: Michael Roth \u003cmdroth@linux.vnet.ibm.com\u003e\n    Signed-off-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Signed-off-by: Sasha Levin \u003csashal@kernel.org\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit b69dab2bd0c9cf508c7cf64a1de10773292fce61\nAuthor: Anay Wadhera \u003canay1018@gmail.com\u003e\nDate:   Sun May 23 18:55:08 2021 +0000\n\n    Revert \"cgroup: Disable IRQs while holding css_set_lock\"\n\n    This reverts commit ac7b270e91c7b0d1b1c5544532852b55177004f1.\n\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 94d828af144997756e881c0d931adc80511ccf40\nAuthor: Colin Cross \u003cccross@android.com\u003e\nDate:   Tue Jul 12 19:53:24 2011 -0700\n\n    cgroup: Add generic cgroup subsystem permission checks\n\n    Rather than using explicit euid \u003d\u003d 0 checks when trying to move\n    tasks into a cgroup via CFS, move permission checks into each\n    specific cgroup subsystem. If a subsystem does not specify a\n    \u0027allow_attach\u0027 handler, then we fall back to doing our checks\n    the old way.\n\n    Use the \u0027allow_attach\u0027 handler for the \u0027cpu\u0027 cgroup to allow\n    non-root processes to add arbitrary processes to a \u0027cpu\u0027 cgroup\n    if it has the CAP_SYS_NICE capability set.\n\n    This version of the patch adds a \u0027allow_attach\u0027 handler instead\n    of reusing the \u0027can_attach\u0027 handler.  If the \u0027can_attach\u0027 handler\n    is reused, a new cgroup that implements \u0027can_attach\u0027 but not\n    the permission checks could end up with no permission checks\n    at all.\n\n    Change-Id: Icfa950aa9321d1ceba362061d32dc7dfa2c64f0c\n    Original-Author: San Mehat \u003csan@google.com\u003e\n    Signed-off-by: Colin Cross \u003cccross@android.com\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 0788f4e1ff2c1d6ef4cd5fe01ae729124fe59337\nAuthor: Rom Lemarchand \u003cromlem@android.com\u003e\nDate:   Fri Nov 7 12:48:17 2014 -0800\n\n    cgroup: refactor allow_attach function into common code\n\n    move cpu_cgroup_allow_attach to a common subsys_cgroup_allow_attach.\n    This allows any process with CAP_SYS_NICE to move tasks across cgroups if\n    they use this function as their allow_attach handler.\n\n    Bug: 18260435\n    Change-Id: I6bb4933d07e889d0dc39e33b4e71320c34a2c90f\n    Signed-off-by: Rom Lemarchand \u003cromlem@android.com\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 79ebef4904c307565e9a19e56c4641e6bc3c0f23\nAuthor: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\nDate:   Fri Jan 22 16:00:56 2021 +0100\n\n    bpf: Fix buggy rsh min/max bounds tracking\n\n    [ no upstream commit ]\n\n    Fix incorrect bounds tracking for RSH opcode. Commit f23cc643f9ba (\"bpf: fix\n    range arithmetic for bpf map access\") had a wrong assumption about min/max\n    bounds. The new dst_reg-\u003emin_value needs to be derived by right shifting the\n    max_val bounds, not min_val, and likewise new dst_reg-\u003emax_value needs to be\n    derived by right shifting the min_val bounds, not max_val. Later stable kernels\n    than 4.9 are not affected since bounds tracking was overall reworked and they\n    already track this similarly as in the fix.\n\n    Fixes: f23cc643f9ba (\"bpf: fix range arithmetic for bpf map access\")\n    Reported-by: Ryota Shiga (Flatt Security)\n    Signed-off-by: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Reviewed-by: John Fastabend \u003cjohn.fastabend@gmail.com\u003e\n    Cc: Josef Bacik \u003cjbacik@fb.com\u003e\n    Signed-off-by: Greg Kroah-Hartman \u003cgregkh@linuxfoundation.org\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit e0985270080a93fc516f8ac07b7fd5c9b073e032\nAuthor: Tejun Heo \u003ctj@kernel.org\u003e\nDate:   Wed Aug 10 11:23:44 2016 -0400\n\n    cgroup: add tracepoints for basic operations\n\n    Debugging what goes wrong with cgroup setup can get hairy.  Add\n    tracepoints for cgroup hierarchy mount, cgroup creation/destruction\n    and task migration operations for better visibility.\n\n    Signed-off-by: Tejun Heo \u003ctj@kernel.org\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 333ba81a68d5c1a99cd9db9ca39fd744343a0d30\nAuthor: Daniel Bristot de Oliveira \u003cbristot@redhat.com\u003e\nDate:   Wed Jun 22 17:28:41 2016 -0300\n\n    cgroup: Disable IRQs while holding css_set_lock\n\n    While testing the deadline scheduler + cgroup setup I hit this\n    warning.\n\n    [  132.612935] ------------[ cut here ]------------\n    [  132.612951] WARNING: CPU: 5 PID: 0 at kernel/softirq.c:150 __local_bh_enable_ip+0x6b/0x80\n    [  132.612952] Modules linked in: (a ton of modules...)\n    [  132.612981] CPU: 5 PID: 0 Comm: swapper/5 Not tainted 4.7.0-rc2 #2\n    [  132.612981] Hardware name: QEMU Standard PC (i440FX + PIIX, 1996), BIOS 1.8.2-20150714_191134- 04/01/2014\n    [  132.612982]  0000000000000086 45c8bb5effdd088b ffff88013fd43da0 ffffffff813d229e\n    [  132.612984]  0000000000000000 0000000000000000 ffff88013fd43de0 ffffffff810a652b\n    [  132.612985]  00000096811387b5 0000000000000200 ffff8800bab29d80 ffff880034c54c00\n    [  132.612986] Call Trace:\n    [  132.612987]  \u003cIRQ\u003e  [\u003cffffffff813d229e\u003e] dump_stack+0x63/0x85\n    [  132.612994]  [\u003cffffffff810a652b\u003e] __warn+0xcb/0xf0\n    [  132.612997]  [\u003cffffffff810e76a0\u003e] ? push_dl_task.part.32+0x170/0x170\n    [  132.612999]  [\u003cffffffff810a665d\u003e] warn_slowpath_null+0x1d/0x20\n    [  132.613000]  [\u003cffffffff810aba5b\u003e] __local_bh_enable_ip+0x6b/0x80\n    [  132.613008]  [\u003cffffffff817d6c8a\u003e] _raw_write_unlock_bh+0x1a/0x20\n    [  132.613010]  [\u003cffffffff817d6c9e\u003e] _raw_spin_unlock_bh+0xe/0x10\n    [  132.613015]  [\u003cffffffff811388ac\u003e] put_css_set+0x5c/0x60\n    [  132.613016]  [\u003cffffffff8113dc7f\u003e] cgroup_free+0x7f/0xa0\n    [  132.613017]  [\u003cffffffff810a3912\u003e] __put_task_struct+0x42/0x140\n    [  132.613018]  [\u003cffffffff810e776a\u003e] dl_task_timer+0xca/0x250\n    [  132.613027]  [\u003cffffffff810e76a0\u003e] ? push_dl_task.part.32+0x170/0x170\n    [  132.613030]  [\u003cffffffff8111371e\u003e] __hrtimer_run_queues+0xee/0x270\n    [  132.613031]  [\u003cffffffff81113ec8\u003e] hrtimer_interrupt+0xa8/0x190\n    [  132.613034]  [\u003cffffffff81051a58\u003e] local_apic_timer_interrupt+0x38/0x60\n    [  132.613035]  [\u003cffffffff817d9b0d\u003e] smp_apic_timer_interrupt+0x3d/0x50\n    [  132.613037]  [\u003cffffffff817d7c5c\u003e] apic_timer_interrupt+0x8c/0xa0\n    [  132.613038]  \u003cEOI\u003e  [\u003cffffffff81063466\u003e] ? native_safe_halt+0x6/0x10\n    [  132.613043]  [\u003cffffffff81037a4e\u003e] default_idle+0x1e/0xd0\n    [  132.613044]  [\u003cffffffff810381cf\u003e] arch_cpu_idle+0xf/0x20\n    [  132.613046]  [\u003cffffffff810e8fda\u003e] default_idle_call+0x2a/0x40\n    [  132.613047]  [\u003cffffffff810e92d7\u003e] cpu_startup_entry+0x2e7/0x340\n    [  132.613048]  [\u003cffffffff81050235\u003e] start_secondary+0x155/0x190\n    [  132.613049] ---[ end trace f91934d162ce9977 ]---\n\n    The warn is the spin_(lock|unlock)_bh(\u0026css_set_lock) in the interrupt\n    context. Converting the spin_lock_bh to spin_lock_irq(save) to avoid\n    this problem - and other problems of sharing a spinlock with an\n    interrupt.\n\n    Cc: Tejun Heo \u003ctj@kernel.org\u003e\n    Cc: Li Zefan \u003clizefan@huawei.com\u003e\n    Cc: Johannes Weiner \u003channes@cmpxchg.org\u003e\n    Cc: Juri Lelli \u003cjuri.lelli@arm.com\u003e\n    Cc: Steven Rostedt \u003crostedt@goodmis.org\u003e\n    Cc: cgroups@vger.kernel.org\n    Cc: stable@vger.kernel.org # 4.5+\n    Cc: linux-kernel@vger.kernel.org\n    Reviewed-by: Rik van Riel \u003criel@redhat.com\u003e\n    Reviewed-by: \"Luis Claudio R. Goncalves\" \u003clgoncalv@redhat.com\u003e\n    Signed-off-by: Daniel Bristot de Oliveira \u003cbristot@redhat.com\u003e\n    Acked-by: Zefan Li \u003clizefan@huawei.com\u003e\n    Signed-off-by: Tejun Heo \u003ctj@kernel.org\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 3417b34c0a062b02b44c663c4dedad65868c2d09\nAuthor: Johannes Weiner \u003channes@cmpxchg.org\u003e\nDate:   Thu Dec 6 09:06:47 2018 -0500\n\n    FROMLIST: kernel: cgroup: add poll file operation\n\n    Cgroup has a standardized poll/notification mechanism for waking all\n    pollers on all fds when a filesystem node changes.  To allow polling for\n    custom events, add a .poll callback that can override the default.\n\n    This is in preparation for pollable cgroup pressure files which have\n    per-fd trigger configurations.\n\n    Link: http://lkml.kernel.org/r/20190124211518.244221-3-surenb@google.com\n    Signed-off-by: Johannes Weiner \u003channes@cmpxchg.org\u003e\n    Signed-off-by: Suren Baghdasaryan \u003csurenb@google.com\u003e\n    Cc: Dennis Zhou \u003cdennis@kernel.org\u003e\n    Cc: Ingo Molnar \u003cmingo@redhat.com\u003e\n    Cc: Jens Axboe \u003caxboe@kernel.dk\u003e\n    Cc: Li Zefan \u003clizefan@huawei.com\u003e\n    Cc: Peter Zijlstra \u003cpeterz@infradead.org\u003e\n    Cc: Tejun Heo \u003ctj@kernel.org\u003e\n    Signed-off-by: Andrew Morton \u003cakpm@linux-foundation.org\u003e\n    Signed-off-by: Stephen Rothwell \u003csfr@canb.auug.org.au\u003e\n\n    (in linux-next: https://git.kernel.org/pub/scm/linux/kernel/git/next/linux-next.git/commit/?id\u003dc88177361203be291a49956b6c9d5ec164ea24b2)\n\n    Conflicts:\n            include/linux/cgroup-defs.h\n            kernel/cgroup.c\n\n    1. made changes in kernel/cgroup.c instead of kernel/cgroup/cgroup.c\n    2. replaced __poll_t with unsigned int\n\n    Bug: 111308141\n    Test: modified lmkd to use PSI and tested using lmkd_unit_test\n\n    Change-Id: Ie3d914197d1f150e1d83c6206865566a7cbff1b4\n    Signed-off-by: Suren Baghdasaryan \u003csurenb@google.com\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 49d2545c922a6a2b61baddde5a85899fa357ba6a\nAuthor: Tejun Heo \u003ctj@kernel.org\u003e\nDate:   Tue Dec 27 14:49:03 2016 -0500\n\n    UPSTREAM: cgroup add cftype-\u003eopen/release() callbacks\n\n    Pipe the newly added kernfs-\u003eopen/release() callbacks through cftype.\n    While at it, as cleanup operations now can be performed from\n    -\u003erelease() instead of -\u003eseq_stop(), make the latter optional.\n\n    Signed-off-by: Tejun Heo \u003ctj@kernel.org\u003e\n    Acked-by: Acked-by: Zefan Li \u003clizefan@huawei.com\u003e\n\n    (cherry picked from commit e90cbebc3fa5caea4c8bfeb0d0157a0cee53efc7)\n\n    Bug: 111308141\n    Test: modified lmkd to use PSI and tested using lmkd_unit_test\n\n    Change-Id: Iff9794cbbc2c7067c24cb2f767bbdeffa26b5180\n    Signed-off-by: Suren Baghdasaryan \u003csurenb@google.com\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 65ef0879b938573a2f2bf1a8c26576847ffb29cd\nAuthor: Zefan Li \u003clizefan@huawei.com\u003e\nDate:   Sat May 9 11:32:10 2020 +0800\n\n    netprio_cgroup: Fix unlimited memory leak of v2 cgroups\n\n    [ Upstream commit 090e28b229af92dc5b40786ca673999d59e73056 ]\n\n    If systemd is configured to use hybrid mode which enables the use of\n    both cgroup v1 and v2, systemd will create new cgroup on both the default\n    root (v2) and netprio_cgroup hierarchy (v1) for a new session and attach\n    task to the two cgroups. If the task does some network thing then the v2\n    cgroup can never be freed after the session exited.\n\n    One of our machines ran into OOM due to this memory leak.\n\n    In the scenario described above when sk_alloc() is called\n    cgroup_sk_alloc() thought it\u0027s in v2 mode, so it stores\n    the cgroup pointer in sk-\u003esk_cgrp_data and increments\n    the cgroup refcnt, but then sock_update_netprioidx()\n    thought it\u0027s in v1 mode, so it stores netprioidx value\n    in sk-\u003esk_cgrp_data, so the cgroup refcnt will never be freed.\n\n    Currently we do the mode switch when someone writes to the ifpriomap\n    cgroup control file. The easiest fix is to also do the switch when\n    a task is attached to a new cgroup.\n\n    Fixes: bd1060a1d671 (\"sock, cgroup: add sock-\u003esk_cgroup\")\n    Reported-by: Yang Yingliang \u003cyangyingliang@huawei.com\u003e\n    Tested-by: Yang Yingliang \u003cyangyingliang@huawei.com\u003e\n    Signed-off-by: Zefan Li \u003clizefan@huawei.com\u003e\n    Acked-by: Tejun Heo \u003ctj@kernel.org\u003e\n    Signed-off-by: Jakub Kicinski \u003ckuba@kernel.org\u003e\n    Signed-off-by: Greg Kroah-Hartman \u003cgregkh@linuxfoundation.org\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 3500f4cf443c8d6aa39519c1191661aa8ebeecbc\nAuthor: Shakeel Butt \u003cshakeelb@google.com\u003e\nDate:   Mon Mar 9 22:16:05 2020 -0700\n\n    cgroup: memcg: net: do not associate sock with unrelated cgroup\n\n    [ Upstream commit e876ecc67db80dfdb8e237f71e5b43bb88ae549c ]\n\n    We are testing network memory accounting in our setup and noticed\n    inconsistent network memory usage and often unrelated cgroups network\n    usage correlates with testing workload. On further inspection, it\n    seems like mem_cgroup_sk_alloc() and cgroup_sk_alloc() are broken in\n    irq context specially for cgroup v1.\n\n    mem_cgroup_sk_alloc() and cgroup_sk_alloc() can be called in irq context\n    and kind of assumes that this can only happen from sk_clone_lock()\n    and the source sock object has already associated cgroup. However in\n    cgroup v1, where network memory accounting is opt-in, the source sock\n    can be unassociated with any cgroup and the new cloned sock can get\n    associated with unrelated interrupted cgroup.\n\n    Cgroup v2 can also suffer if the source sock object was created by\n    process in the root cgroup or if sk_alloc() is called in irq context.\n    The fix is to just do nothing in interrupt.\n\n    WARNING: Please note that about half of the TCP sockets are allocated\n    from the IRQ context, so, memory used by such sockets will not be\n    accouted by the memcg.\n\n    The stack trace of mem_cgroup_sk_alloc() from IRQ-context:\n\n    CPU: 70 PID: 12720 Comm: ssh Tainted:  5.6.0-smp-DEV #1\n    Hardware name: ...\n    Call Trace:\n     \u003cIRQ\u003e\n     dump_stack+0x57/0x75\n     mem_cgroup_sk_alloc+0xe9/0xf0\n     sk_clone_lock+0x2a7/0x420\n     inet_csk_clone_lock+0x1b/0x110\n     tcp_create_openreq_child+0x23/0x3b0\n     tcp_v6_syn_recv_sock+0x88/0x730\n     tcp_check_req+0x429/0x560\n     tcp_v6_rcv+0x72d/0xa40\n     ip6_protocol_deliver_rcu+0xc9/0x400\n     ip6_input+0x44/0xd0\n     ? ip6_protocol_deliver_rcu+0x400/0x400\n     ip6_rcv_finish+0x71/0x80\n     ipv6_rcv+0x5b/0xe0\n     ? ip6_sublist_rcv+0x2e0/0x2e0\n     process_backlog+0x108/0x1e0\n     net_rx_action+0x26b/0x460\n     __do_softirq+0x104/0x2a6\n     do_softirq_own_stack+0x2a/0x40\n     \u003c/IRQ\u003e\n     do_softirq.part.19+0x40/0x50\n     __local_bh_enable_ip+0x51/0x60\n     ip6_finish_output2+0x23d/0x520\n     ? ip6table_mangle_hook+0x55/0x160\n     __ip6_finish_output+0xa1/0x100\n     ip6_finish_output+0x30/0xd0\n     ip6_output+0x73/0x120\n     ? __ip6_finish_output+0x100/0x100\n     ip6_xmit+0x2e3/0x600\n     ? ipv6_anycast_cleanup+0x50/0x50\n     ? inet6_csk_route_socket+0x136/0x1e0\n     ? skb_free_head+0x1e/0x30\n     inet6_csk_xmit+0x95/0xf0\n     __tcp_transmit_skb+0x5b4/0xb20\n     __tcp_send_ack.part.60+0xa3/0x110\n     tcp_send_ack+0x1d/0x20\n     tcp_rcv_state_process+0xe64/0xe80\n     ? tcp_v6_connect+0x5d1/0x5f0\n     tcp_v6_do_rcv+0x1b1/0x3f0\n     ? tcp_v6_do_rcv+0x1b1/0x3f0\n     __release_sock+0x7f/0xd0\n     release_sock+0x30/0xa0\n     __inet_stream_connect+0x1c3/0x3b0\n     ? prepare_to_wait+0xb0/0xb0\n     inet_stream_connect+0x3b/0x60\n     __sys_connect+0x101/0x120\n     ? __sys_getsockopt+0x11b/0x140\n     __x64_sys_connect+0x1a/0x20\n     do_syscall_64+0x51/0x200\n     entry_SYSCALL_64_after_hwframe+0x44/0xa9\n\n    The stack trace of mem_cgroup_sk_alloc() from IRQ-context:\n    Fixes: 2d7580738345 (\"mm: memcontrol: consolidate cgroup socket tracking\")\n    Fixes: d979a39d7242 (\"cgroup: duplicate cgroup reference when cloning sockets\")\n    Signed-off-by: Shakeel Butt \u003cshakeelb@google.com\u003e\n    Reviewed-by: Roman Gushchin \u003cguro@fb.com\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Greg Kroah-Hartman \u003cgregkh@linuxfoundation.org\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 9e585103bce276b4db0b6a130d72dcacd1d9ab20\nAuthor: Yang Yingliang \u003cyangyingliang@huawei.com\u003e\nDate:   Thu Aug 13 20:27:57 2020 +0000\n\n    cgroup: add missing skcd-\u003eno_refcnt check in cgroup_sk_clone()\n\n    Add skcd-\u003eno_refcnt check which is missed when backporting\n    ad0f75e5f57c (\"cgroup: fix cgroup_sk_alloc() for sk_clone_lock()\").\n\n    This patch is needed in stable-4.9, stable-4.14 and stable-4.19.\n\n    Signed-off-by: Yang Yingliang \u003cyangyingliang@huawei.com\u003e\n    Signed-off-by: Sasha Levin \u003csashal@kernel.org\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 2c5b51f1bb6b73cf14ece028a035329e6a1d748e\nAuthor: Cong Wang \u003cxiyou.wangcong@gmail.com\u003e\nDate:   Thu Jul 2 11:52:56 2020 -0700\n\n    cgroup: fix cgroup_sk_alloc() for sk_clone_lock()\n\n    [ Upstream commit ad0f75e5f57ccbceec13274e1e242f2b5a6397ed ]\n\n    When we clone a socket in sk_clone_lock(), its sk_cgrp_data is\n    copied, so the cgroup refcnt must be taken too. And, unlike the\n    sk_alloc() path, sock_update_netprioidx() is not called here.\n    Therefore, it is safe and necessary to grab the cgroup refcnt\n    even when cgroup_sk_alloc is disabled.\n\n    sk_clone_lock() is in BH context anyway, the in_interrupt()\n    would terminate this function if called there. And for sk_alloc()\n    skcd-\u003eval is always zero. So it\u0027s safe to factor out the code\n    to make it more readable.\n\n    The global variable \u0027cgroup_sk_alloc_disabled\u0027 is used to determine\n    whether to take these reference counts. It is impossible to make\n    the reference counting correct unless we save this bit of information\n    in skcd-\u003eval. So, add a new bit there to record whether the socket\n    has already taken the reference counts. This obviously relies on\n    kmalloc() to align cgroup pointers to at least 4 bytes,\n    ARCH_KMALLOC_MINALIGN is certainly larger than that.\n\n    This bug seems to be introduced since the beginning, commit\n    d979a39d7242 (\"cgroup: duplicate cgroup reference when cloning sockets\")\n    tried to fix it but not compeletely. It seems not easy to trigger until\n    the recent commit 090e28b229af\n    (\"netprio_cgroup: Fix unlimited memory leak of v2 cgroups\") was merged.\n\n    Fixes: bd1060a1d671 (\"sock, cgroup: add sock-\u003esk_cgroup\")\n    Reported-by: Cameron Berkenpas \u003ccam@neo-zeon.de\u003e\n    Reported-by: Peter Geis \u003cpgwipeout@gmail.com\u003e\n    Reported-by: Lu Fengqi \u003clufq.fnst@cn.fujitsu.com\u003e\n    Reported-by: Daniël Sonck \u003cdsonck92@gmail.com\u003e\n    Reported-by: Zhang Qiang \u003cqiang.zhang@windriver.com\u003e\n    Tested-by: Cameron Berkenpas \u003ccam@neo-zeon.de\u003e\n    Tested-by: Peter Geis \u003cpgwipeout@gmail.com\u003e\n    Tested-by: Thomas Lamprecht \u003ct.lamprecht@proxmox.com\u003e\n    Cc: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Cc: Zefan Li \u003clizefan@huawei.com\u003e\n    Cc: Tejun Heo \u003ctj@kernel.org\u003e\n    Cc: Roman Gushchin \u003cguro@fb.com\u003e\n    Signed-off-by: Cong Wang \u003cxiyou.wangcong@gmail.com\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Greg Kroah-Hartman \u003cgregkh@linuxfoundation.org\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 0769a838c5c8049ec876f703f2f5812ae1881066\nAuthor: Chenbo Feng \u003cfengc@google.com\u003e\nDate:   Wed Mar 22 17:27:35 2017 -0700\n\n    BACKPORT: UPSTREAM: Add a eBPF helper function to retrieve socket uid\n\n    Cherry-pick from commit 6acc5c2910689fc6ee181bf63085c5efff6a42bd\n\n    Returns the owner uid of the socket inside a sk_buff. This is useful to\n    perform per-UID accounting of network traffic or per-UID packet\n    filtering. The socket need to be a fullsock otherwise overflowuid is\n    returned.\n\n    Signed-off-by: Chenbo Feng \u003cfengc@google.com\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Bug: 30950746\n    Change-Id: Idc00947ccfdd4e9f2214ffc4178d701cd9ead0ac\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit b042c02048e1a9d9860c409ed638d61e6f36cb9c\nAuthor: Chenbo Feng \u003cfengc@google.com\u003e\nDate:   Wed Mar 22 17:27:34 2017 -0700\n\n    BACKPORT: UPSTREAM: Add a helper function to get socket cookie in eBPF\n\n    Cherrypick from commit: 91b8270f2a4d1d9b268de90451cdca63a70052d6\n\n    Retrieve the socket cookie generated by sock_gen_cookie() from a sk_buff\n    with a known socket. Generates a new cookie if one was not yet set.If\n    the socket pointer inside sk_buff is NULL, 0 is returned. The helper\n    function coud be useful in monitoring per socket networking traffic\n    statistics and provide a unique socket identifier per namespace.\n\n    Acked-by: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Acked-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Acked-by: Willem de Bruijn \u003cwillemb@google.com\u003e\n    Signed-off-by: Chenbo Feng \u003cfengc@google.com\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Bug: 30950746\n    Change-Id: I95918dcc3ceffb3061495a859d28aee88e3cde3c\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit d37171aecd9d5fc669d3811a0283d594957bda2d\nAuthor: Chenbo Feng \u003cfengc@google.com\u003e\nDate:   Wed May 3 15:22:42 2017 -0700\n\n    ANDROID: Fix missing uapi headers\n\n    Update the missing bpf helper function name in bpf_func_id to keep the\n    uapi header consistent with upstream uapi header because we need the\n    new added bpf helper function bpf get_socket_cookie and get_socket_uid.\n    The patch related to those headers are not backetported since they are\n    not related and backport them will bring in extra confilict.\n\n    Signed-off-by: Chenbo Feng \u003cfengc@google.com\u003e\n    Bug: 30950746\n    Change-Id: I2b5fd03799ac5f2e3243ab11a1bccb932f06c312\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 2b5c5bedb9701ba1fcb8e54067a856aeb103fb98\nAuthor: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\nDate:   Fri Sep 23 01:28:37 2016 +0200\n\n    bpf: add helper to invalidate hash\n\n    Add a small helper that complements 36bbef52c7eb (\"bpf: direct packet\n    write and access for helpers for clsact progs\") for invalidating the\n    current skb-\u003ehash after mangling on headers via direct packet write.\n\n    Signed-off-by: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Acked-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 8f8fdc7db3b78616b38a12c53e2ae2944fb944f4\nAuthor: Anay Wadhera \u003cawadhera@berkeley.edu\u003e\nDate:   Fri May 21 15:18:35 2021 -0700\n\n    net: take compile fix from 4.9\n\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 8138272a39a5a3ee29a65b1816f3216d882f9f95\nAuthor: Anay Wadhera \u003cawadhera@berkeley.edu\u003e\nDate:   Fri May 21 15:06:45 2021 -0700\n\n    cgroup: replace out_idr_free with actual code\n\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 2aad4250db090889c674ffd3b3ec1583158a0c96\nAuthor: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\nDate:   Wed Mar 9 03:00:02 2016 +0100\n\n    ip_tunnel: add support for setting flow label via collect metadata\n\n    This patch extends udp_tunnel6_xmit_skb() to pass in the IPv6 flow label\n    from call sites. Currently, there\u0027s no such option and it\u0027s always set to\n    zero when writing ip6_flow_hdr(). Add a label member to ip_tunnel_key, so\n    that flow-based tunnels via collect metadata frontends can make use of it.\n    vxlan and geneve will be converted to add flow label support separately.\n\n    Signed-off-by: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 34c6c1f1c1a32901c4db911ae2d29ed9778b4b1b\nAuthor: Jamal Hadi Salim \u003cjhs@mojatatu.com\u003e\nDate:   Sat Jul 2 06:43:14 2016 -0400\n\n    net: simplify and make pkt_type_ok() available for other users\n\n    Suggested-by: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Signed-off-by: Jamal Hadi Salim \u003cjhs@mojatatu.com\u003e\n    Acked-by: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit f37b41508db5d2dc8ea95938d337a6cd109fddec\nAuthor: Aditya Kali \u003cadityakali@google.com\u003e\nDate:   Fri Jan 29 02:54:08 2016 -0600\n\n    kernfs: define kernfs_node_dentry\n\n    Add a new kernfs api is added to lookup the dentry for a particular\n    kernfs path.\n\n    Signed-off-by: Aditya Kali \u003cadityakali@google.com\u003e\n    Signed-off-by: Serge E. Hallyn \u003cserge.hallyn@canonical.com\u003e\n    Acked-by: Greg Kroah-Hartman \u003cgregkh@linuxfoundation.org\u003e\n    Signed-off-by: Tejun Heo \u003ctj@kernel.org\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 868300a337fd8c73f5b6a87192bf7b8a0cdae675\nAuthor: Tejun Heo \u003ctj@kernel.org\u003e\nDate:   Fri Aug 11 05:49:01 2017 -0700\n\n    BACKPORT: cgroup: misc changes\n\n    Misc trivial changes to prepare for future changes.  No functional\n    difference.\n\n    * Expose cgroup_get(), cgroup_tryget() and cgroup_parent().\n\n    * Implement task_dfl_cgroup() which dereferences css_set-\u003edfl_cgrp.\n\n    * Rename cgroup_stats_show() to cgroup_stat_show() for consistency\n      with the file name.\n\n    Signed-off-by: Tejun Heo \u003ctj@kernel.org\u003e\n\n    (cherry picked from commit 3e48930cc74f0c212ee1838f89ad0ca7fcf2fea1)\n\n    Conflicts:\n            kernel/cgroup/cgroup.c\n\n    (1. manual merge because kernel/cgroup/cgroup.c is under kernel/cgroup.c\n    2. cgroup_stats_show change is skipped because the function dos not exist)\n\n    Bug: 111308141\n    Test: modified lmkd to use PSI and tested using lmkd_unit_test\n\n    Signed-off-by: Suren Baghdasaryan \u003csurenb@google.com\u003e\n    Change-Id: I756ee3dcf0d0f3da69cd1b58e644271625053538\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit ad8b6378eb7539914b444aa63a1d8f8665e1749e\nAuthor: Josh Poimboeuf \u003cjpoimboe@redhat.com\u003e\nDate:   Wed Mar 1 12:04:44 2017 -0600\n\n    objtool, modules: Discard objtool annotation sections for modules\n\n    commit e390f9a9689a42f477a6073e2e7df530a4c1b740 upstream.\n\n    The \u0027__unreachable\u0027 and \u0027__func_stack_frame_non_standard\u0027 sections are\n    only used at compile time.  They\u0027re discarded for vmlinux but they\n    should also be discarded for modules.\n\n    Since this is a recurring pattern, prefix the section names with\n    \".discard.\".  It\u0027s a nice convention and vmlinux.lds.h already discards\n    such sections.\n\n    Also remove the \u0027a\u0027 (allocatable) flag from the __unreachable section\n    since it doesn\u0027t make sense for a discarded section.\n\n    Suggested-by: Linus Torvalds \u003ctorvalds@linux-foundation.org\u003e\n    Signed-off-by: Josh Poimboeuf \u003cjpoimboe@redhat.com\u003e\n    Cc: Jessica Yu \u003cjeyu@redhat.com\u003e\n    Cc: Peter Zijlstra \u003cpeterz@infradead.org\u003e\n    Cc: Thomas Gleixner \u003ctglx@linutronix.de\u003e\n    Fixes: d1091c7fa3d5 (\"objtool: Improve detection of BUG() and other dead ends\")\n    Link: http://lkml.kernel.org/r/20170301180444.lhd53c5tibc4ns77@treble\n    Signed-off-by: Ingo Molnar \u003cmingo@kernel.org\u003e\n    [dwmw2: Remove the unreachable part in backporting since it\u0027s not here yet]\n    Signed-off-by: David Woodhouse \u003cdwmw@amazon.co.ku\u003e\n    Signed-off-by: Greg Kroah-Hartman \u003cgregkh@linuxfoundation.org\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 7875074852deaeb63000514a606b385bccac5ffd\nAuthor: Josh Poimboeuf \u003cjpoimboe@redhat.com\u003e\nDate:   Sun Feb 28 22:22:35 2016 -0600\n\n    objtool: Add STACK_FRAME_NON_STANDARD() macro\n\n    Add a new macro, STACK_FRAME_NON_STANDARD(), which is used to denote a\n    function which does something unusual related to its stack frame.  Use\n    of the macro prevents objtool from emitting a false positive warning.\n\n    Signed-off-by: Josh Poimboeuf \u003cjpoimboe@redhat.com\u003e\n    Cc: Andrew Morton \u003cakpm@linux-foundation.org\u003e\n    Cc: Andy Lutomirski \u003cluto@kernel.org\u003e\n    Cc: Arnaldo Carvalho de Melo \u003cacme@kernel.org\u003e\n    Cc: Bernd Petrovitsch \u003cbernd@petrovitsch.priv.at\u003e\n    Cc: Borislav Petkov \u003cbp@alien8.de\u003e\n    Cc: Chris J Arges \u003cchris.j.arges@canonical.com\u003e\n    Cc: Jiri Slaby \u003cjslaby@suse.cz\u003e\n    Cc: Linus Torvalds \u003ctorvalds@linux-foundation.org\u003e\n    Cc: Michal Marek \u003cmmarek@suse.cz\u003e\n    Cc: Namhyung Kim \u003cnamhyung@gmail.com\u003e\n    Cc: Pedro Alves \u003cpalves@redhat.com\u003e\n    Cc: Peter Zijlstra \u003cpeterz@infradead.org\u003e\n    Cc: Thomas Gleixner \u003ctglx@linutronix.de\u003e\n    Cc: live-patching@vger.kernel.org\n    Link: http://lkml.kernel.org/r/34487a17b23dba43c50941599d47054a9584b219.1456719558.git.jpoimboe@redhat.com\n    Signed-off-by: Ingo Molnar \u003cmingo@kernel.org\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 2c3ed942b62c8c552b90af9ec5f2e33fe6f8e263\nAuthor: Anay Wadhera \u003cawadhera@berkeley.edu\u003e\nDate:   Fri May 21 14:45:03 2021 -0700\n\n    remove leftovers from 6ea07b4590d3174a53303\n\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 7a851110e4899b9d6e0f4b2765131f086671017d\nAuthor: Eric W. Biederman \u003cebiederm@xmission.com\u003e\nDate:   Tue May 24 09:29:01 2016 -0500\n\n    fs: Add user namespace member to struct super_block\n\n    Start marking filesystems with a user namespace owner, s_user_ns.  In\n    this change this is only used for permission checks of who may mount a\n    filesystem.  Ultimately s_user_ns will be used for translating ids and\n    checking capabilities for filesystems mounted from user namespaces.\n\n    The default policy for setting s_user_ns is implemented in sget(),\n    which arranges for s_user_ns to be set to current_user_ns() and to\n    ensure that the mounter of the filesystem has CAP_SYS_ADMIN in that\n    user_ns.\n\n    The guts of sget are split out into another function sget_userns().\n    The function sget_userns calls alloc_super with the specified user\n    namespace or it verifies the existing superblock that was found\n    has the expected user namespace, and fails with EBUSY when it is not.\n    This failing prevents users with the wrong privileges mounting a\n    filesystem.\n\n    The reason for the split of sget_userns from sget is that in some\n    cases such as mount_ns and kernfs_mount_ns a different policy for\n    permission checking of mounts and setting s_user_ns is necessary, and\n    the existence of sget_userns() allows those policies to be\n    implemented.\n\n    The helper mount_ns is expected to be used for filesystems such as\n    proc and mqueuefs which present per namespace information.  The\n    function mount_ns is modified to call sget_userns instead of sget to\n    ensure the user namespace owner of the namespace whose information is\n    presented by the filesystem is used on the superblock.\n\n    For sysfs and cgroup the appropriate permission checks are already in\n    place, and kernfs_mount_ns is modified to call sget_userns so that\n    the init_user_ns is the only user namespace used.\n\n    For the cgroup filesystem cgroup namespace mounts are bind mounts of a\n    subset of the full cgroup filesystem and as such s_user_ns must be the\n    same for all of them as there is only a single superblock.\n\n    Mounts of sysfs that vary based on the network namespace could in principle\n    change s_user_ns but it keeps the analysis and implementation of kernfs\n    simpler if that is not supported, and at present there appear to be no\n    benefits from supporting a different s_user_ns on any sysfs mount.\n\n    Getting the details of setting s_user_ns correct has been\n    a long process.  Thanks to Pavel Tikhorirorv who spotted a leak\n    in sget_userns.  Thanks to Seth Forshee who has kept the work alive.\n\n    Thanks-to: Seth Forshee \u003cseth.forshee@canonical.com\u003e\n    Thanks-to: Pavel Tikhomirov \u003cptikhomirov@virtuozzo.com\u003e\n    Acked-by: Seth Forshee \u003cseth.forshee@canonical.com\u003e\n    Signed-off-by: Eric W. Biederman \u003cebiederm@xmission.com\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 0c4c71d19f69a92aa67affe258ca3aa3741ad31e\nAuthor: Eric W. Biederman \u003cebiederm@xmission.com\u003e\nDate:   Mon May 23 14:51:59 2016 -0500\n\n    vfs: Pass data, ns, and ns-\u003euserns to mount_ns\n\n    Today what is normally called data (the mount options) is not passed\n    to fill_super through mount_ns.\n\n    Pass the mount options and the namespace separately to mount_ns so\n    that filesystems such as proc that have mount options, can use\n    mount_ns.\n\n    Pass the user namespace to mount_ns so that the standard permission\n    check that verifies the mounter has permissions over the namespace can\n    be performed in mount_ns instead of in each filesystems .mount method.\n    Thus removing the duplication between mqueuefs and proc in terms of\n    permission checks.  The extra permission check does not currently\n    affect the rpc_pipefs filesystem and the nfsd filesystem as those\n    filesystems do not currently allow unprivileged mounts.  Without\n    unpvileged mounts it is guaranteed that the caller has already passed\n    capable(CAP_SYS_ADMIN) which guarantees extra permission check will\n    pass.\n\n    Update rpc_pipefs and the nfsd filesystem to ensure that the network\n    namespace reference is always taken in fill_super and always put in kill_sb\n    so that the logic is simpler and so that errors originating inside of\n    fill_super do not cause a network namespace leak.\n\n    Acked-by: Seth Forshee \u003cseth.forshee@canonical.com\u003e\n    Signed-off-by: \"Eric W. Biederman\" \u003cebiederm@xmission.com\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 1c1dc68943c9d6d5a181081e470ba4b1022f9ef4\nAuthor: Tejun Heo \u003ctj@kernel.org\u003e\nDate:   Wed Aug 10 11:23:44 2016 -0400\n\n    kernfs: make kernfs_path*() behave in the style of strlcpy()\n\n    kernfs_path*() functions always return the length of the full path but\n    the path content is undefined if the length is larger than the\n    provided buffer.  This makes its behavior different from strlcpy() and\n    requires error handling in all its users even when they don\u0027t care\n    about truncation.  In addition, the implementation can actully be\n    simplified by making it behave properly in strlcpy() style.\n\n    * Update kernfs_path_from_node_locked() to always fill up the buffer\n      with path.  If the buffer is not large enough, the output is\n      truncated and terminated.\n\n    * kernfs_path() no longer needs error handling.  Make it a simple\n      inline wrapper around kernfs_path_from_node().\n\n    * sysfs_warn_dup()\u0027s use of kernfs_path() doesn\u0027t need error handling.\n      Updated accordingly.\n\n    * cgroup_path()\u0027s use of kernfs_path() updated to retain the old\n      behavior.\n\n    Signed-off-by: Tejun Heo \u003ctj@kernel.org\u003e\n    Acked-by: Greg Kroah-Hartman \u003cgregkh@linuxfoundation.org\u003e\n    Acked-by: Serge Hallyn \u003cserge.hallyn@ubuntu.com\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 2d96590ea44abecaabbf9f070cff5a8c8fa6d12f\nAuthor: Serge Hallyn \u003cserge.hallyn@ubuntu.com\u003e\nDate:   Sun Apr 17 15:04:31 2016 -0500\n\n    kernfs_path_from_node_locked: don\u0027t overwrite nlen\n\n    We\u0027ve calculated @len to be the bytes we need for \u0027/..\u0027 entries from\n    @kn_from to the common ancestor, and calculated @nlen to be the extra\n    bytes we need to get from the common ancestor to @kn_to.  We use them\n    as such at the end.  But in the loop copying the actual entries, we\n    overwrite @nlen.  Use a temporary variable for that instead.\n\n    Without this, the return length, when the buffer is large enough, is\n    wrong.  (When the buffer is NULL or too small, the returned value is\n    correct. The buffer contents are also correct.)\n\n    Interestingly, no callers of this function are affected by this as of\n    yet.  However the upcoming cgroup_show_path() will be.\n\n    Signed-off-by: Serge Hallyn \u003cserge.hallyn@ubuntu.com\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 9d93b38a76d5d9173ef34f36569f749cd6a2a697\nAuthor: Anay Wadhera \u003cawadhera@berkeley.edu\u003e\nDate:   Fri May 21 14:29:50 2021 -0700\n\n    arm64: bpf_jit_comp: drop artifact\n\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 594aeac0e18946504f1888fd54c68d3d8f0f8ada\nAuthor: Arnd Bergmann \u003carnd@arndb.de\u003e\nDate:   Tue Jan 10 13:08:06 2017 +0100\n\n    UPSTREAM: cgroup: move CONFIG_SOCK_CGROUP_DATA to init/Kconfig\n\n    We now \u0027select SOCK_CGROUP_DATA\u0027 but Kconfig complains that this is\n    not right when CONFIG_NET is disabled and there is no socket interface:\n\n    warning: (CGROUP_BPF) selects SOCK_CGROUP_DATA which has unmet direct dependencies (NET)\n\n    I don\u0027t know what the correct solution for this is, but simply removing\n    the dependency on NET from SOCK_CGROUP_DATA by moving it out of the\n    \u0027if NET\u0027 section avoids the warning and does not produce other build\n    errors.\n\n    Fixes: 483c4933ea09 (\"cgroup: Fix CGROUP_BPF config\")\n    Signed-off-by: Arnd Bergmann \u003carnd@arndb.de\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n\n    Fixes: Change-Id: Ib41ef78fba02eb9e592558ddbf06f9ec0aa337b6\n           (\"UPSTREAM: cgroup: Fix CGROUP_BPF config\")\n    (cherry picked from commit 73b351473547e543e9c8166dd67fd99c64c15b0b)\n    Signed-off-by: Amit Pundir \u003camit.pundir@linaro.org\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 13efe07f3d1f009cd2c2babe6169447585bd32aa\nAuthor: Andy Lutomirski \u003cluto@kernel.org\u003e\nDate:   Fri Dec 16 08:33:45 2016 -0800\n\n    UPSTREAM: cgroup: Fix CGROUP_BPF config\n\n    Cherry-pick from commit 483c4933ea09b7aa625b9d64af286fc22ec7e419\n\n    CGROUP_BPF depended on SOCK_CGROUP_DATA which can\u0027t be manually\n    enabled, making it rather challenging to turn CGROUP_BPF on.\n\n    Signed-off-by: Andy Lutomirski \u003cluto@kernel.org\u003e\n    Acked-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Bug: 30950746\n    Change-Id: Ib41ef78fba02eb9e592558ddbf06f9ec0aa337b6\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 210d70ce9fb35bdec6fb41defa232ffef6383dd4\nAuthor: Yonghong Song \u003cyhs@fb.com\u003e\nDate:   Mon Oct 23 23:53:08 2017 -0700\n\n    BACKPORT: bpf: permit multiple bpf attachments for a single perf event\n\n    This patch enables multiple bpf attachments for a\n    kprobe/uprobe/tracepoint single trace event.\n    Each trace_event keeps a list of attached perf events.\n    When an event happens, all attached bpf programs will\n    be executed based on the order of attachment.\n\n    A global bpf_event_mutex lock is introduced to protect\n    prog_array attaching and detaching. An alternative will\n    be introduce a mutex lock in every trace_event_call\n    structure, but it takes a lot of extra memory.\n    So a global bpf_event_mutex lock is a good compromise.\n\n    The bpf prog detachment involves allocation of memory.\n    If the allocation fails, a dummy do-nothing program\n    will replace to-be-detached program in-place.\n\n    Signed-off-by: Yonghong Song \u003cyhs@fb.com\u003e\n    Acked-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Acked-by: Martin KaFai Lau \u003ckafai@fb.com\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    (cherry picked from commit e87c6bc3852b981e71c757be20771546ce9f76f3)\n    Signed-off-by: Connor O\u0027Brien \u003cconnoro@google.com\u003e\n    Bug: 121213201\n    Bug: 138317270\n    Test: build \u0026 boot cuttlefish; attach 2 progs to 1 tracepoint\n    Change-Id: I390d8c0146888ddb1aed5a6f6e5dae7ef394ebc9\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 171b5880e473c4bb6bbe4ac54bd7446aa403119e\nAuthor: Alexei Starovoitov \u003cast@fb.com\u003e\nDate:   Mon Apr 18 20:11:50 2016 -0700\n\n    perf, bpf: minimize the size of perf_trace_() tracepoint handler\n\n    move trace_call_bpf() into helper function to minimize the size\n    of perf_trace_*() tracepoint handlers.\n        text\t   data\t    bss\t    dec\t \t   hex\tfilename\n    10541679\t5526646\t2945024\t19013349\t1221ee5\tvmlinux_before\n    10509422\t5526646\t2945024\t18981092\t121a0e4\tvmlinux_after\n\n    It may seem that perf_fetch_caller_regs() can also be moved,\n    but that is incorrect, since ip/sp will be wrong.\n\n    bpf+tracepoint performance is not affected, since\n    perf_swevent_put_recursion_context() is now inlined.\n    export_symbol_gpl can also be dropped.\n\n    No measurable change in normal perf tracepoints.\n\n    Suggested-by: Steven Rostedt \u003crostedt@goodmis.org\u003e\n    Signed-off-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Acked-by: Peter Zijlstra (Intel) \u003cpeterz@infradead.org\u003e\n    Acked-by: Steven Rostedt \u003crostedt@goodmis.org\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 01ff66a68246938515f9380cf0199f85ba2de95b\nAuthor: Yonghong Song \u003cyhs@fb.com\u003e\nDate:   Mon Oct 23 23:53:07 2017 -0700\n\n    UPSTREAM: bpf: use the same condition in perf event set/free bpf handler\n\n    This is a cleanup such that doing the same check in\n    perf_event_free_bpf_prog as we already do in\n    perf_event_set_bpf_prog step.\n\n    Signed-off-by: Yonghong Song \u003cyhs@fb.com\u003e\n    Acked-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Acked-by: Martin KaFai Lau \u003ckafai@fb.com\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    (cherry picked from commit 0b4c6841fee03e096b735074a0c4aab3a8e92986)\n    Signed-off-by: Connor O\u0027Brien \u003cconnoro@google.com\u003e\n    Bug: 121213201\n    Bug: 138317270\n    Test: build \u0026 boot cuttlefish\n    Change-Id: Ie423e73a73be29e8ef50cc22dbb03e14e241c8de\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 48644c6700cfe80a2773a1e89740face00ca8906\nAuthor: Alexei Starovoitov \u003cast@fb.com\u003e\nDate:   Mon Oct 2 22:50:21 2017 -0700\n\n    BACKPORT: bpf: multi program support for cgroup+bpf\n\n    introduce BPF_F_ALLOW_MULTI flag that can be used to attach multiple\n    bpf programs to a cgroup.\n\n    The difference between three possible flags for BPF_PROG_ATTACH command:\n    - NONE(default): No further bpf programs allowed in the subtree.\n    - BPF_F_ALLOW_OVERRIDE: If a sub-cgroup installs some bpf program,\n      the program in this cgroup yields to sub-cgroup program.\n    - BPF_F_ALLOW_MULTI: If a sub-cgroup installs some bpf program,\n      that cgroup program gets run in addition to the program in this cgroup.\n\n    NONE and BPF_F_ALLOW_OVERRIDE existed before. This patch doesn\u0027t\n    change their behavior. It only clarifies the semantics in relation\n    to new flag.\n\n    Only one program is allowed to be attached to a cgroup with\n    NONE or BPF_F_ALLOW_OVERRIDE flag.\n    Multiple programs are allowed to be attached to a cgroup with\n    BPF_F_ALLOW_MULTI flag. They are executed in FIFO order\n    (those that were attached first, run first)\n    The programs of sub-cgroup are executed first, then programs of\n    this cgroup and then programs of parent cgroup.\n    All eligible programs are executed regardless of return code from\n    earlier programs.\n\n    To allow efficient execution of multiple programs attached to a cgroup\n    and to avoid penalizing cgroups without any programs attached\n    introduce \u0027struct bpf_prog_array\u0027 which is RCU protected array\n    of pointers to bpf programs.\n\n    Signed-off-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Acked-by: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Acked-by: Martin KaFai Lau \u003ckafai@fb.com\u003e\n    for cgroup bits\n    Acked-by: Tejun Heo \u003ctj@kernel.org\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    (cherry picked from commit 324bda9e6c5add86ba2e1066476481c48132aca0)\n    Signed-off-by: Connor O\u0027Brien \u003cconnoro@google.com\u003e\n    Bug: 121213201\n    Bug: 138317270\n    Test: build \u0026 boot cuttlefish\n    Change-Id: I06b71c850b9f3e052b106abab7a4a3add012a3f8\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit ec820a2fad007bb6691c60787ffe1f83f31149c4\nAuthor: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\nDate:   Sat Aug 17 00:00:08 2019 +0100\n\n    bpf: add bpf_jit_limit knob to restrict unpriv allocations\n\n    commit ede95a63b5e84ddeea6b0c473b36ab8bfd8c6ce3 upstream.\n\n    Rick reported that the BPF JIT could potentially fill the entire module\n    space with BPF programs from unprivileged users which would prevent later\n    attempts to load normal kernel modules or privileged BPF programs, for\n    example. If JIT was enabled but unsuccessful to generate the image, then\n    before commit 290af86629b2 (\"bpf: introduce BPF_JIT_ALWAYS_ON config\")\n    we would always fall back to the BPF interpreter. Nowadays in the case\n    where the CONFIG_BPF_JIT_ALWAYS_ON could be set, then the load will abort\n    with a failure since the BPF interpreter was compiled out.\n\n    Add a global limit and enforce it for unprivileged users such that in case\n    of BPF interpreter compiled out we fail once the limit has been reached\n    or we fall back to BPF interpreter earlier w/o using module mem if latter\n    was compiled in. In a next step, fair share among unprivileged users can\n    be resolved in particular for the case where we would fail hard once limit\n    is reached.\n\n    Fixes: 290af86629b2 (\"bpf: introduce BPF_JIT_ALWAYS_ON config\")\n    Fixes: 0a14842f5a3c (\"net: filter: Just In Time compiler for x86-64\")\n    Co-Developed-by: Rick Edgecombe \u003crick.p.edgecombe@intel.com\u003e\n    Signed-off-by: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Acked-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Cc: Eric Dumazet \u003ceric.dumazet@gmail.com\u003e\n    Cc: Jann Horn \u003cjannh@google.com\u003e\n    Cc: Kees Cook \u003ckeescook@chromium.org\u003e\n    Cc: LKML \u003clinux-kernel@vger.kernel.org\u003e\n    Signed-off-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    [bwh: Backported to 4.9: adjust context]\n    Signed-off-by: Ben Hutchings \u003cben.hutchings@codethink.co.uk\u003e\n    Signed-off-by: Greg Kroah-Hartman \u003cgregkh@linuxfoundation.org\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 94a01e3257754b5265545453f1b27444a2ee10b3\nAuthor: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\nDate:   Fri Aug 16 23:59:56 2019 +0100\n\n    bpf: restrict access to core bpf sysctls\n\n    commit 2e4a30983b0f9b19b59e38bbf7427d7fdd480d98 upstream.\n\n    Given BPF reaches far beyond just networking these days, it was\n    never intended to allow setting and in some cases reading those\n    knobs out of a user namespace root running without CAP_SYS_ADMIN,\n    thus tighten such access.\n\n    Also the bpf_jit_enable \u003d 2 debugging mode should only be allowed\n    if kptr_restrict is not set since it otherwise can leak addresses\n    to the kernel log. Dump a note to the kernel log that this is for\n    debugging JITs only when enabled.\n\n    Signed-off-by: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Acked-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Signed-off-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    [bwh: Backported to 4.9:\n     - We don\u0027t have bpf_dump_raw_ok(), so drop the condition based on it. This\n       condition only made it a bit harder for a privileged user to do something\n       silly.\n     - Drop change to bpf_jit_kallsyms]\n    Signed-off-by: Ben Hutchings \u003cben.hutchings@codethink.co.uk\u003e\n    Signed-off-by: Greg Kroah-Hartman \u003cgregkh@linuxfoundation.org\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 1f20c7e5888e71b46d15831d7a42c36d084147a2\nAuthor: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\nDate:   Fri Aug 16 23:59:20 2019 +0100\n\n    bpf: get rid of pure_initcall dependency to enable jits\n\n    commit fa9dd599b4dae841924b022768354cfde9affecb upstream.\n\n    Having a pure_initcall() callback just to permanently enable BPF\n    JITs under CONFIG_BPF_JIT_ALWAYS_ON is unnecessary and could leave\n    a small race window in future where JIT is still disabled on boot.\n    Since we know about the setting at compilation time anyway, just\n    initialize it properly there. Also consolidate all the individual\n    bpf_jit_enable variables into a single one and move them under one\n    location. Moreover, don\u0027t allow for setting unspecified garbage\n    values on them.\n\n    Signed-off-by: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Acked-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Signed-off-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    [bwh: Backported to 4.9 as dependency of commit 2e4a30983b0f\n     \"bpf: restrict access to core bpf sysctls\":\n     - Drop change in arch/mips/net/ebpf_jit.c\n     - Drop change to bpf_jit_kallsyms\n     - Adjust filenames, context]\n    Signed-off-by: Ben Hutchings \u003cben.hutchings@codethink.co.uk\u003e\n    Signed-off-by: Greg Kroah-Hartman \u003cgregkh@linuxfoundation.org\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit ad818aeb27f38149ea00ff050b5aefb6c7dedb6c\nAuthor: Zi Shen Lim \u003czlim.lnx@gmail.com\u003e\nDate:   Wed Jun 8 21:18:48 2016 -0700\n\n    arm64: bpf: implement bpf_tail_call() helper\n\n    Add support for JMP_CALL_X (tail call) introduced by commit 04fd61ab36ec\n    (\"bpf: allow bpf programs to tail-call other bpf programs\").\n\n    bpf_tail_call() arguments:\n      ctx   - context pointer passed to next program\n      array - pointer to map which type is BPF_MAP_TYPE_PROG_ARRAY\n      index - index inside array that selects specific program to run\n\n    In this implementation arm64 JIT jumps into callee program after prologue,\n    so callee program reuses the same stack. For tail_call_cnt, we use the\n    callee-saved R26 (which was already saved/restored but previously unused\n    by JIT).\n\n    With this patch a tail call generates the following code on arm64:\n\n      if (index \u003e\u003d array-\u003emap.max_entries)\n          goto out;\n\n      34:   mov     x10, #0x10                      // #16\n      38:   ldr     w10, [x1,x10]\n      3c:   cmp     w2, w10\n      40:   b.ge    0x0000000000000074\n\n      if (tail_call_cnt \u003e MAX_TAIL_CALL_CNT)\n          goto out;\n      tail_call_cnt++;\n\n      44:   mov     x10, #0x20                      // #32\n      48:   cmp     x26, x10\n      4c:   b.gt    0x0000000000000074\n      50:   add     x26, x26, #0x1\n\n      prog \u003d array-\u003eptrs[index];\n      if (prog \u003d\u003d NULL)\n          goto out;\n\n      54:   mov     x10, #0x68                      // #104\n      58:   ldr     x10, [x1,x10]\n      5c:   ldr     x11, [x10,x2]\n      60:   cbz     x11, 0x0000000000000074\n\n      goto *(prog-\u003ebpf_func + prologue_size);\n\n      64:   mov     x10, #0x20                      // #32\n      68:   ldr     x10, [x11,x10]\n      6c:   add     x10, x10, #0x20\n      70:   br      x10\n      74:\n\n    Signed-off-by: Zi Shen Lim \u003czlim.lnx@gmail.com\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 14de6c12580533008343d6da357fc8285ce23a3c\nAuthor: Yang Shi \u003cyang.shi@linaro.org\u003e\nDate:   Mon May 16 16:36:26 2016 -0700\n\n    bpf: arm64: remove callee-save registers use for tmp registers\n\n    In the current implementation of ARM64 eBPF JIT, R23 and R24 are used for\n    tmp registers, which are callee-saved registers. This leads to variable size\n    of JIT prologue and epilogue. The latest blinding constant change prefers to\n    constant size of prologue and epilogue. AAPCS reserves R9 ~ R15 for temp\n    registers which not need to be saved/restored during function call. So, replace\n    R23 and R24 to R10 and R11, and remove tmp_used flag to save 2 instructions for\n    some jited BPF program.\n\n    CC: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Acked-by: Zi Shen Lim \u003czlim.lnx@gmail.com\u003e\n    Signed-off-by: Yang Shi \u003cyang.shi@linaro.org\u003e\n    Acked-by: Catalin Marinas \u003ccatalin.marinas@arm.com\u003e\n    Acked-by: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit cd6d8cc255e3396e4957cc452160fa4ac3b0188f\nAuthor: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\nDate:   Fri May 13 19:08:34 2016 +0200\n\n    bpf, arm64: add support for constant blinding\n\n    This patch adds recently added constant blinding helpers into the\n    arm64 eBPF JIT. In the bpf_int_jit_compile() path, requirements are\n    to utilize bpf_jit_blind_constants()/bpf_jit_prog_release_other()\n    pair for rewriting the program into a blinded one, and to map the\n    BPF_REG_AX register to a CPU register. The mapping is on x9.\n\n    Signed-off-by: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Acked-by: Zi Shen Lim \u003czlim.lnx@gmail.com\u003e\n    Acked-by: Yang Shi \u003cyang.shi@linaro.org\u003e\n    Tested-by: Yang Shi \u003cyang.shi@linaro.org\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit c8d6591e9094a942a2d6a1773fdd971343baf70c\nAuthor: Zi Shen Lim \u003czlim.lnx@gmail.com\u003e\nDate:   Wed Jan 13 23:33:22 2016 -0800\n\n    arm64: bpf: add extra pass to handle faulty codegen\n\n    Code generation functions in arch/arm64/kernel/insn.c previously\n    BUG_ON invalid parameters. Following change of that behavior, now we\n    need to handle the error case where AARCH64_BREAK_FAULT is returned.\n\n    Instead of error-handling on every emit() in JIT, we add a new\n    validation pass at the end of JIT compilation. There\u0027s no point in\n    running JITed code at run-time only to trap due to AARCH64_BREAK_FAULT.\n    Instead, we drop this failed JIT compilation and allow the system to\n    gracefully fallback on the BPF interpreter.\n\n    Signed-off-by: Zi Shen Lim \u003czlim.lnx@gmail.com\u003e\n    Suggested-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Acked-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 61a2502d89c0f7434196da6f413fb2ea2d0de5ba\nAuthor: Valdis Klētnieks \u003cvaldis.kletnieks@vt.edu\u003e\nDate:   Thu Jun 6 22:39:27 2019 -0400\n\n    bpf: silence warning messages in core\n\n    [ Upstream commit aee450cbe482a8c2f6fa5b05b178ef8b8ff107ca ]\n\n    Compiling kernel/bpf/core.c with W\u003d1 causes a flood of warnings:\n\n    kernel/bpf/core.c:1198:65: warning: initialized field overwritten [-Woverride-init]\n     1198 | #define BPF_INSN_3_TBL(x, y, z) [BPF_##x | BPF_##y | BPF_##z] \u003d true\n          |                                                                 ^~~~\n    kernel/bpf/core.c:1087:2: note: in expansion of macro \u0027BPF_INSN_3_TBL\u0027\n     1087 |  INSN_3(ALU, ADD,  X),   \\\n          |  ^~~~~~\n    kernel/bpf/core.c:1202:3: note: in expansion of macro \u0027BPF_INSN_MAP\u0027\n     1202 |   BPF_INSN_MAP(BPF_INSN_2_TBL, BPF_INSN_3_TBL),\n          |   ^~~~~~~~~~~~\n    kernel/bpf/core.c:1198:65: note: (near initialization for \u0027public_insntable[12]\u0027)\n     1198 | #define BPF_INSN_3_TBL(x, y, z) [BPF_##x | BPF_##y | BPF_##z] \u003d true\n          |                                                                 ^~~~\n    kernel/bpf/core.c:1087:2: note: in expansion of macro \u0027BPF_INSN_3_TBL\u0027\n     1087 |  INSN_3(ALU, ADD,  X),   \\\n          |  ^~~~~~\n    kernel/bpf/core.c:1202:3: note: in expansion of macro \u0027BPF_INSN_MAP\u0027\n     1202 |   BPF_INSN_MAP(BPF_INSN_2_TBL, BPF_INSN_3_TBL),\n          |   ^~~~~~~~~~~~\n\n    98 copies of the above.\n\n    The attached patch silences the warnings, because we *know* we\u0027re overwriting\n    the default initializer. That leaves bpf/core.c with only 6 other warnings,\n    which become more visible in comparison.\n\n    Signed-off-by: Valdis Kletnieks \u003cvaldis.kletnieks@vt.edu\u003e\n    Acked-by: Andrii Nakryiko \u003candriin@fb.com\u003e\n    Signed-off-by: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Signed-off-by: Sasha Levin \u003csashal@kernel.org\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 14ab33fe132b7e94832ded176597663e91b42c41\nAuthor: Chenbo Feng \u003cfengc@google.com\u003e\nDate:   Tue May 14 19:42:57 2019 -0700\n\n    UPSTREAM: bpf: relax inode permission check for retrieving bpf program\n\n    For iptable module to load a bpf program from a pinned location, it\n    only retrieve a loaded program and cannot change the program content so\n    requiring a write permission for it might not be necessary.\n    Also when adding or removing an unrelated iptable rule, it might need to\n    flush and reload the xt_bpf related rules as well and triggers the inode\n    permission check. It might be better to remove the write premission\n    check for the inode so we won\u0027t need to grant write access to all the\n    processes that flush and restore iptables rules.\n\n    Signed-off-by: Chenbo Feng \u003cfengc@google.com\u003e\n    Signed-off-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    (cherry picked from commit e547ff3f803e779a3898f1f48447b29f43c54085)\n\n    Bug: 129650054\n    Change-Id: I71487ad6f4d22e0a8be3757d9b72d1c04c92104d\n    (cherry picked from commit 9e74c1b9e8418aa0209b15db24f0b3d4876f52aa)\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit fa40b3355684c092ebdca700d04bd1de7d09e0c8\nAuthor: Alexei Starovoitov \u003cast@fb.com\u003e\nDate:   Thu May 9 19:33:54 2019 -0700\n\n    bpf: convert htab map to hlist_nulls\n\n    commit 4fe8435909fddc97b81472026aa954e06dd192a5 upstream.\n\n    when all map elements are pre-allocated one cpu can delete and reuse htab_elem\n    while another cpu is still walking the hlist. In such case the lookup may\n    miss the element. Convert hlist to hlist_nulls to avoid such scenario.\n    When bucket lock is taken there is no need to take such precautions,\n    so only convert map_lookup and map_get_next to nulls.\n    The race window is extremely small and only reproducible with explicit\n    udelay() inside lookup_nulls_elem_raw()\n\n    Similar to hlist add hlist_nulls_for_each_entry_safe() and\n    hlist_nulls_entry_safe() helpers.\n\n    Fixes: 6c9059817432 (\"bpf: pre-allocate hash map elements\")\n    Reported-by: Jonathan Perry \u003cjonperry@fb.com\u003e\n    Signed-off-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Acked-by: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Chenbo Feng \u003cfengc@google.com\u003e\n    Signed-off-by: Sasha Levin \u003csashal@kernel.org\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 0e39f5284b0a5c0424d1c446b5bd2e228cf69d86\nAuthor: Alexei Starovoitov \u003cast@fb.com\u003e\nDate:   Thu May 9 19:33:53 2019 -0700\n\n    bpf: fix struct htab_elem layout\n\n    commit 9f691549f76d488a0c74397b3e51e943865ea01f upstream.\n\n    when htab_elem is removed from the bucket list the htab_elem.hash_node.next\n    field should not be overridden too early otherwise we have a tiny race window\n    between lookup and delete.\n    The bug was discovered by manual code analysis and reproducible\n    only with explicit udelay() in lookup_elem_raw().\n\n    Fixes: 6c9059817432 (\"bpf: pre-allocate hash map elements\")\n    Reported-by: Jonathan Perry \u003cjonperry@fb.com\u003e\n    Signed-off-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Acked-by: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Chenbo Feng \u003cfengc@google.com\u003e\n    Signed-off-by: Sasha Levin \u003csashal@kernel.org\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit b2b555f2f28ff308ff373e78aeca0bb8c1a14ea9\nAuthor: Alexei Starovoitov \u003cast@kernel.org\u003e\nDate:   Mon Dec 3 22:46:04 2018 -0800\n\n    bpf: check pending signals while verifying programs\n\n    [ Upstream commit c3494801cd1785e2c25f1a5735fa19ddcf9665da ]\n\n    Malicious user space may try to force the verifier to use as much cpu\n    time and memory as possible. Hence check for pending signals\n    while verifying the program.\n    Note that suspend of sys_bpf(PROG_LOAD) syscall will lead to EAGAIN,\n    since the kernel has to release the resources used for program verification.\n\n    Reported-by: Anatoly Trosinenko \u003canatoly.trosinenko@gmail.com\u003e\n    Signed-off-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Acked-by: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Acked-by: Edward Cree \u003cecree@solarflare.com\u003e\n    Signed-off-by: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Signed-off-by: Sasha Levin \u003csashal@kernel.org\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 2e3c1d2b56458ea9d89ed6a4e2004b6771d9f779\nAuthor: Alexei Starovoitov \u003cast@kernel.org\u003e\nDate:   Tue May 15 09:27:05 2018 -0700\n\n    bpf: Prevent memory disambiguation attack\n\n    commit af86ca4e3088fe5eacf2f7e58c01fa68ca067672 upstream.\n\n    Detect code patterns where malicious \u0027speculative store bypass\u0027 can be used\n    and sanitize such patterns.\n\n     39: (bf) r3 \u003d r10\n     40: (07) r3 +\u003d -216\n     41: (79) r8 \u003d *(u64 *)(r7 +0)   // slow read\n     42: (7a) *(u64 *)(r10 -72) \u003d 0  // verifier inserts this instruction\n     43: (7b) *(u64 *)(r8 +0) \u003d r3   // this store becomes slow due to r8\n     44: (79) r1 \u003d *(u64 *)(r6 +0)   // cpu speculatively executes this load\n     45: (71) r2 \u003d *(u8 *)(r1 +0)    // speculatively arbitrary \u0027load byte\u0027\n                                     // is now sanitized\n\n    Above code after x86 JIT becomes:\n     e5: mov    %rbp,%rdx\n     e8: add    $0xffffffffffffff28,%rdx\n     ef: mov    0x0(%r13),%r14\n     f3: movq   $0x0,-0x48(%rbp)\n     fb: mov    %rdx,0x0(%r14)\n     ff: mov    0x0(%rbx),%rdi\n    103: movzbq 0x0(%rdi),%rsi\n\n    Signed-off-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Signed-off-by: Thomas Gleixner \u003ctglx@linutronix.de\u003e\n    [bwh: Backported to 4.9:\n     - Add bpf_verifier_env parameter to check_stack_write()\n     - Look up stack slot_types with state-\u003estack_slot_type[] rather than\n       state-\u003estack[].slot_type[]\n     - Drop bpf_verifier_env argument to verbose()\n     - Adjust context]\n    Signed-off-by: Ben Hutchings \u003cben.hutchings@codethink.co.uk\u003e\n    Signed-off-by: Greg Kroah-Hartman \u003cgregkh@linuxfoundation.org\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 30fa864349c6f77795e6ae6149752c0b3a724918\nAuthor: Ben Hutchings \u003cben.hutchings@codethink.co.uk\u003e\nDate:   Wed Dec 5 22:41:36 2018 +0000\n\n    bpf/verifier: Pass instruction index to check_mem_access() and check_xadd()\n\n    Extracted from commit 31fd85816dbe \"bpf: permits narrower load from\n    bpf program context fields\".\n\n    Cc: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Cc: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Signed-off-by: Ben Hutchings \u003cben.hutchings@codethink.co.uk\u003e\n    Signed-off-by: Greg Kroah-Hartman \u003cgregkh@linuxfoundation.org\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit c130ccf583d4f037e36666232aa804c4ac6fe41d\nAuthor: Ben Hutchings \u003cben.hutchings@codethink.co.uk\u003e\nDate:   Wed Dec 5 22:45:15 2018 +0000\n\n    bpf/verifier: Add spi variable to check_stack_write()\n\n    Extracted from commit dc503a8ad984 \"bpf/verifier: track liveness for\n    pruning\".\n\n    Cc: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Cc: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Signed-off-by: Ben Hutchings \u003cben.hutchings@codethink.co.uk\u003e\n    Signed-off-by: Greg Kroah-Hartman \u003cgregkh@linuxfoundation.org\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 2656b2b16ffb89c4d55e82e7de2f070555b9f836\nAuthor: Jakub Kicinski \u003cjakub.kicinski@netronome.com\u003e\nDate:   Thu May 3 18:37:17 2018 -0700\n\n    bpf: fix references to free_bpf_prog_info() in comments\n\n    [ Upstream commit ab7f5bf0928be2f148d000a6eaa6c0a36e74750e ]\n\n    Comments in the verifier refer to free_bpf_prog_info() which\n    seems to have never existed in tree.  Replace it with\n    free_used_maps().\n\n    Signed-off-by: Jakub Kicinski \u003cjakub.kicinski@netronome.com\u003e\n    Reviewed-by: Quentin Monnet \u003cquentin.monnet@netronome.com\u003e\n    Signed-off-by: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Signed-off-by: Sasha Levin \u003calexander.levin@microsoft.com\u003e\n    Signed-off-by: Greg Kroah-Hartman \u003cgregkh@linuxfoundation.org\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 425f0ce763baf8ca5a8120cd99348fa8c5633d83\nAuthor: Teng Qin \u003cqinteng@fb.com\u003e\nDate:   Mon Apr 24 19:00:37 2017 -0700\n\n    bpf: map_get_next_key to return first key on NULL\n\n    commit 8fe45924387be6b5c1be59a7eb330790c61d5d10 upstream.\n\n    When iterating through a map, we need to find a key that does not exist\n    in the map so map_get_next_key will give us the first key of the map.\n    This often requires a lot of guessing in production systems.\n\n    This patch makes map_get_next_key return the first key when the key\n    pointer in the parameter is NULL.\n\n    Signed-off-by: Teng Qin \u003cqinteng@fb.com\u003e\n    Signed-off-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Acked-by: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Chenbo Feng \u003cfengc@google.com\u003e\n    Cc: Lorenzo Colitti \u003clorenzo@google.com\u003e\n    Signed-off-by: Greg Kroah-Hartman \u003cgregkh@linuxfoundation.org\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit ebbedefd2db167856fea5315ecf0f0e4102631aa\nAuthor: Chenbo Feng \u003cfengc@google.com\u003e\nDate:   Mon Mar 19 17:57:27 2018 -0700\n\n    bpf: skip unnecessary capability check\n\n    commit 0fa4fe85f4724fff89b09741c437cbee9cf8b008 upstream.\n\n    The current check statement in BPF syscall will do a capability check\n    for CAP_SYS_ADMIN before checking sysctl_unprivileged_bpf_disabled. This\n    code path will trigger unnecessary security hooks on capability checking\n    and cause false alarms on unprivileged process trying to get CAP_SYS_ADMIN\n    access. This can be resolved by simply switch the order of the statement\n    and CAP_SYS_ADMIN is not required anyway if unprivileged bpf syscall is\n    allowed.\n\n    Signed-off-by: Chenbo Feng \u003cfengc@google.com\u003e\n    Acked-by: Lorenzo Colitti \u003clorenzo@google.com\u003e\n    Signed-off-by: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Signed-off-by: Greg Kroah-Hartman \u003cgregkh@linuxfoundation.org\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 27d0cafea1fd6570a6ca341d8dee178db6f985ad\nAuthor: Al Viro \u003cviro@zeniv.linux.org.uk\u003e\nDate:   Sat Dec 2 20:20:38 2017 -0500\n\n    BACKPORT: fix \"netfilter: xt_bpf: Fix XT_BPF_MODE_FD_PINNED mode of \u0027xt_bpf_info_v1\u0027\"\n\n    Descriptor table is a shared object; it\u0027s not a place where you can\n    stick temporary references to files, especially when we don\u0027t need\n    an opened file at all.\n\n    Cc: stable@vger.kernel.org # v4.14\n    Fixes: 98589a0998b8 (\"netfilter: xt_bpf: Fix XT_BPF_MODE_FD_PINNED mode of \u0027xt_bpf_info_v1\u0027\")\n    Signed-off-by: Al Viro \u003cviro@zeniv.linux.org.uk\u003e\n    Signed-off-by: Chenbo Feng \u003cfengc@google.com\u003e\n\n    Removed the code related to function bpf_prog_get_ok() since it is not\n    exsit in current android tree.\n    (cherry picked from commit 040ee69226f8a96b7943645d68f41d5d44b5ff7d)\n\n    Change-Id: If7a602128cdea4b4b50c8effb215c9bca7449515\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit eb13760b2d095d7c6cdc8db5c9ad71c110d90177\nAuthor: Jann Horn \u003cjannh@google.com\u003e\nDate:   Fri Dec 1 01:46:07 2017 +0100\n\n    UPSTREAM: netfilter: xt_bpf: add overflow checks\n\n    Check whether inputs from userspace are too long (explicit length field too\n    big or string not null-terminated) to avoid out-of-bounds reads.\n\n    As far as I can tell, this can at worst lead to very limited kernel heap\n    memory disclosure or oopses.\n\n    This bug can be triggered by an unprivileged user even if the xt_bpf module\n    is not loaded: iptables is available in network namespaces, and the xt_bpf\n    module can be autoloaded.\n\n    Triggering the bug with a classic BPF filter with fake length 0x1000 causes\n    the following KASAN report:\n\n    \u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\n    BUG: KASAN: slab-out-of-bounds in bpf_prog_create+0x84/0xf0\n    Read of size 32768 at addr ffff8801eff2c494 by task test/4627\n\n    CPU: 0 PID: 4627 Comm: test Not tainted 4.15.0-rc1+ #1\n    [...]\n    Call Trace:\n     dump_stack+0x5c/0x85\n     print_address_description+0x6a/0x260\n     kasan_report+0x254/0x370\n     ? bpf_prog_create+0x84/0xf0\n     memcpy+0x1f/0x50\n     bpf_prog_create+0x84/0xf0\n     bpf_mt_check+0x90/0xd6 [xt_bpf]\n    [...]\n    Allocated by task 4627:\n     kasan_kmalloc+0xa0/0xd0\n     __kmalloc_node+0x47/0x60\n     xt_alloc_table_info+0x41/0x70 [x_tables]\n    [...]\n    The buggy address belongs to the object at ffff8801eff2c3c0\n                    which belongs to the cache kmalloc-2048 of size 2048\n    The buggy address is located 212 bytes inside of\n                    2048-byte region [ffff8801eff2c3c0, ffff8801eff2cbc0)\n    [...]\n    \u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\n\n    Fixes: e6f30c731718 (\"netfilter: x_tables: add xt_bpf match\")\n    Signed-off-by: Jann Horn \u003cjannh@google.com\u003e\n    Signed-off-by: Pablo Neira Ayuso \u003cpablo@netfilter.org\u003e\n    Signed-off-by: Chenbo Feng \u003cfengc@google.com\u003e\n\n    (cherry picked from commit 6ab405114b0b229151ef06f4e31c7834dd09d0c0)\n\n    Change-Id: Ie066a9df84812853a9c9d2e51aa53646f4001542\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 195a85f0ceb365221f5d7845cfdda9591aaf1029\nAuthor: Shmulik Ladkani \u003cshmulik.ladkani@gmail.com\u003e\nDate:   Mon Oct 9 15:27:15 2017 +0300\n\n    UPSTREAM: netfilter: xt_bpf: Fix XT_BPF_MODE_FD_PINNED mode of \u0027xt_bpf_info_v1\u0027\n\n    Commit 2c16d6033264 (\"netfilter: xt_bpf: support ebpf\") introduced\n    support for attaching an eBPF object by an fd, with the\n    \u0027bpf_mt_check_v1\u0027 ABI expecting the \u0027.fd\u0027 to be specified upon each\n    IPT_SO_SET_REPLACE call.\n\n    However this breaks subsequent iptables calls:\n\n     # iptables -A INPUT -m bpf --object-pinned /sys/fs/bpf/xxx -j ACCEPT\n     # iptables -A INPUT -s 5.6.7.8 -j ACCEPT\n     iptables: Invalid argument. Run `dmesg\u0027 for more information.\n\n    That\u0027s because iptables works by loading existing rules using\n    IPT_SO_GET_ENTRIES to userspace, then issuing IPT_SO_SET_REPLACE with\n    the replacement set.\n\n    However, the loaded \u0027xt_bpf_info_v1\u0027 has an arbitrary \u0027.fd\u0027 number\n    (from the initial \"iptables -m bpf\" invocation) - so when 2nd invocation\n    occurs, userspace passes a bogus fd number, which leads to\n    \u0027bpf_mt_check_v1\u0027 to fail.\n\n    One suggested solution [1] was to hack iptables userspace, to perform a\n    \"entries fixup\" immediatley after IPT_SO_GET_ENTRIES, by opening a new,\n    process-local fd per every \u0027xt_bpf_info_v1\u0027 entry seen.\n\n    However, in [2] both Pablo Neira Ayuso and Willem de Bruijn suggested to\n    depricate the xt_bpf_info_v1 ABI dealing with pinned ebpf objects.\n\n    This fix changes the XT_BPF_MODE_FD_PINNED behavior to ignore the given\n    \u0027.fd\u0027 and instead perform an in-kernel lookup for the bpf object given\n    the provided \u0027.path\u0027.\n\n    It also defines an alias for the XT_BPF_MODE_FD_PINNED mode, named\n    XT_BPF_MODE_PATH_PINNED, to better reflect the fact that the user is\n    expected to provide the path of the pinned object.\n\n    Existing XT_BPF_MODE_FD_ELF behavior (non-pinned fd mode) is preserved.\n\n    References: [1] https://marc.info/?l\u003dnetfilter-devel\u0026m\u003d150564724607440\u0026w\u003d2\n                [2] https://marc.info/?l\u003dnetfilter-devel\u0026m\u003d150575727129880\u0026w\u003d2\n\n    Reported-by: Rafael Buchbinder \u003crafi@rbk.ms\u003e\n    Signed-off-by: Shmulik Ladkani \u003cshmulik.ladkani@gmail.com\u003e\n    Acked-by: Willem de Bruijn \u003cwillemb@google.com\u003e\n    Acked-by: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Signed-off-by: Pablo Neira Ayuso \u003cpablo@netfilter.org\u003e\n    Signed-off-by: Chenbo Feng \u003cfengc@google.com\u003e\n    (cherry picked from commit 98589a0998b8b13c4a8fa1ccb0e62751a019faa5)\n\n    Change-Id: Ia0d15a76823cca3afb38786a3d2c25c13ccf941d\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 4ba7b554f50c90a09d5ddc8bbdea6554906a5b1a\nAuthor: Willem de Bruijn \u003cwillemb@google.com\u003e\nDate:   Tue Dec 6 16:25:02 2016 -0500\n\n    UPSTREAM: netfilter: xt_bpf: support ebpf\n\n    Add support for attaching an eBPF object by file descriptor.\n\n    The iptables binary can be called with a path to an elf object or a\n    pinned bpf object. Also pass the mode and path to the kernel to be\n    able to return it later for iptables dump and save.\n\n    Signed-off-by: Willem de Bruijn \u003cwillemb@google.com\u003e\n    Signed-off-by: Pablo Neira Ayuso \u003cpablo@netfilter.org\u003e\n    Signed-off-by: Chenbo Feng \u003cfengc@google.com\u003e\n    (cherry picked from commit 2c16d60332643e90d4fa244f4a706c454b8c7569)\n\n    Change-Id: I31b8831a7ffd7c44985ee906ff194c1d934dafbe\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit ec42eb1f5122368db93dfcc53d6e6f56e26835cc\nAuthor: Alexei Starovoitov \u003cast@fb.com\u003e\nDate:   Thu Sep 1 18:37:24 2016 -0700\n\n    perf, bpf: add perf events core support for BPF_PROG_TYPE_PERF_EVENT programs\n\n    Allow attaching BPF_PROG_TYPE_PERF_EVENT programs to sw and hw perf events\n    via overflow_handler mechanism.\n    When program is attached the overflow_handlers become stacked.\n    The program acts as a filter.\n    Returning zero from the program means that the normal perf_event_output handler\n    will not be called and sampling event won\u0027t be stored in the ring buffer.\n\n    The overflow_handler_context\u003d\u003dNULL is an additional safety check\n    to make sure programs are not attached to hw breakpoints and watchdog\n    in case other checks (that prevent that now anyway) get accidentally\n    relaxed in the future.\n\n    The program refcnt is incremented in case perf_events are inhereted\n    when target task is forked.\n    Similar to kprobe and tracepoint programs there is no ioctl to\n    detach the program or swap already attached program. The user space\n    expected to close(perf_event_fd) like it does right now for kprobe+bpf.\n    That restriction simplifies the code quite a bit.\n\n    The invocation of overflow_handler in __perf_event_overflow() is now\n    done via READ_ONCE, since that pointer can be replaced when the program\n    is attached while perf_event itself could have been active already.\n    There is no need to do similar treatment for event-\u003eprog, since it\u0027s\n    assigned only once before it\u0027s accessed.\n\n    Signed-off-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit f3a8a6f4b22c6887e9fdf4522f86be6ef1e53f85\nAuthor: Wang Nan \u003cwangnan0@huawei.com\u003e\nDate:   Mon Mar 28 06:41:30 2016 +0000\n\n    perf/core: Set event\u0027s default ::overflow_handler()\n\n    Set a default event-\u003eoverflow_handler in perf_event_alloc() so don\u0027t\n    need to check event-\u003eoverflow_handler in __perf_event_overflow().\n    Following commits can give a different default overflow_handler.\n\n    Initial idea comes from Peter:\n\n      http://lkml.kernel.org/r/20130708121557.GA17211@twins.programming.kicks-ass.net\n\n    Since the default value of event-\u003eoverflow_handler is not NULL, existing\n    \u0027if (!overflow_handler)\u0027 checks need to be changed.\n\n    is_default_overflow_handler() is introduced for this.\n\n    No extra performance overhead is introduced into the hot path because in the\n    original code we still need to read this handler from memory. A conditional\n    branch is avoided so actually we remove some instructions.\n\n    Signed-off-by: Wang Nan \u003cwangnan0@huawei.com\u003e\n    Signed-off-by: Peter Zijlstra (Intel) \u003cpeterz@infradead.org\u003e\n    Cc: \u003cpi3orama@163.com\u003e\n    Cc: Alexander Shishkin \u003calexander.shishkin@linux.intel.com\u003e\n    Cc: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Cc: Arnaldo Carvalho de Melo \u003cacme@redhat.com\u003e\n    Cc: Brendan Gregg \u003cbrendan.d.gregg@gmail.com\u003e\n    Cc: He Kuang \u003chekuang@huawei.com\u003e\n    Cc: Jiri Olsa \u003cjolsa@kernel.org\u003e\n    Cc: Jiri Olsa \u003cjolsa@redhat.com\u003e\n    Cc: Linus Torvalds \u003ctorvalds@linux-foundation.org\u003e\n    Cc: Masami Hiramatsu \u003cmasami.hiramatsu.pt@hitachi.com\u003e\n    Cc: Namhyung Kim \u003cnamhyung@kernel.org\u003e\n    Cc: Peter Zijlstra \u003cpeterz@infradead.org\u003e\n    Cc: Stephane Eranian \u003ceranian@google.com\u003e\n    Cc: Thomas Gleixner \u003ctglx@linutronix.de\u003e\n    Cc: Vince Weaver \u003cvincent.weaver@maine.edu\u003e\n    Cc: Zefan Li \u003clizefan@huawei.com\u003e\n    Link: http://lkml.kernel.org/r/1459147292-239310-3-git-send-email-wangnan0@huawei.com\n    Signed-off-by: Ingo Molnar \u003cmingo@kernel.org\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 55be5db96c1bb48c44b5a2018e4c409b7d844b8b\nAuthor: Eric Dumazet \u003cedumazet@google.com\u003e\nDate:   Thu Mar 8 16:17:36 2018 +0100\n\n    bpf: add schedule points in percpu arrays management\n\n    [ upstream commit 32fff239de37ef226d5b66329dd133f64d63b22d ]\n\n    syszbot managed to trigger RCU detected stalls in\n    bpf_array_free_percpu()\n\n    It takes time to allocate a huge percpu map, but even more time to free\n    it.\n\n    Since we run in process context, use cond_resched() to yield cpu if\n    needed.\n\n    Fixes: a10423b87a7e (\"bpf: introduce BPF_MAP_TYPE_PERCPU_ARRAY map\")\n    Signed-off-by: Eric Dumazet \u003cedumazet@google.com\u003e\n    Reported-by: syzbot \u003csyzkaller@googlegroups.com\u003e\n    Signed-off-by: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Signed-off-by: Greg Kroah-Hartman \u003cgregkh@linuxfoundation.org\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit baf0ea99179eb7891d735a333370fbbc07f34db7\nAuthor: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\nDate:   Thu Mar 8 16:17:33 2018 +0100\n\n    bpf: fix mlock precharge on arraymaps\n\n    [ upstream commit 9c2d63b843a5c8a8d0559cc067b5398aa5ec3ffc ]\n\n    syzkaller recently triggered OOM during percpu map allocation;\n    while there is work in progress by Dennis Zhou to add __GFP_NORETRY\n    semantics for percpu allocator under pressure, there seems also a\n    missing bpf_map_precharge_memlock() check in array map allocation.\n\n    Given today the actual bpf_map_charge_memlock() happens after the\n    find_and_alloc_map() in syscall path, the bpf_map_precharge_memlock()\n    is there to bail out early before we go and do the map setup work\n    when we find that we hit the limits anyway. Therefore add this for\n    array map as well.\n\n    Fixes: 6c9059817432 (\"bpf: pre-allocate hash map elements\")\n    Fixes: a10423b87a7e (\"bpf: introduce BPF_MAP_TYPE_PERCPU_ARRAY map\")\n    Reported-by: syzbot+adb03f3f0bb57ce3acda@syzkaller.appspotmail.com\n    Signed-off-by: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Cc: Dennis Zhou \u003cdennisszhou@gmail.com\u003e\n    Signed-off-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Signed-off-by: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Signed-off-by: Greg Kroah-Hartman \u003cgregkh@linuxfoundation.org\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 2830ede9ba5ac6ebdca7a72e11fceb5711bd9d13\nAuthor: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\nDate:   Thu Mar 8 16:17:32 2018 +0100\n\n    bpf: fix wrong exposure of map_flags into fdinfo for lpm\n\n    [ upstream commit a316338cb71a3260201490e615f2f6d5c0d8fb2c ]\n\n    trie_alloc() always needs to have BPF_F_NO_PREALLOC passed in via\n    attr-\u003emap_flags, since it does not support preallocation yet. We\n    check the flag, but we never copy the flag into trie-\u003emap.map_flags,\n    which is later on exposed into fdinfo and used by loaders such as\n    iproute2. Latter uses this in bpf_map_selfcheck_pinned() to test\n    whether a pinned map has the same spec as the one from the BPF obj\n    file and if not, bails out, which is currently the case for lpm\n    since it exposes always 0 as flags.\n\n    Also copy over flags in array_map_alloc() and stack_map_alloc().\n    They always have to be 0 right now, but we should make sure to not\n    miss to copy them over at a later point in time when we add actual\n    flags for them to use.\n\n    Fixes: b95a5c4db09b (\"bpf: add a longest prefix match trie map implementation\")\n    Reported-by: Jarno Rajahalme \u003cjarno@covalent.io\u003e\n    Signed-off-by: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Acked-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Signed-off-by: Greg Kroah-Hartman \u003cgregkh@linuxfoundation.org\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 76afe46462fd92157dc93e4b4b896f673a81e6e0\nAuthor: Sami Tolvanen \u003csamitolvanen@google.com\u003e\nDate:   Thu Aug 24 08:59:31 2017 -0700\n\n    bpf: fix function type for __bpf_prog_run\n\n    Bug: 67506682\n    Change-Id: I096a470c65a2a1867c51da9a33843ae23bf5e547\n    Signed-off-by: Sami Tolvanen \u003csamitolvanen@google.com\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 31e0ee4629891c304a7367ed0eccadab700a0945\nAuthor: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\nDate:   Mon Jan 29 02:49:01 2018 +0100\n\n    bpf: reject stores into ctx via st and xadd\n\n    [ upstream commit f37a8cb84cce18762e8f86a70bd6a49a66ab964c ]\n\n    Alexei found that verifier does not reject stores into context\n    via BPF_ST instead of BPF_STX. And while looking at it, we\n    also should not allow XADD variant of BPF_STX.\n\n    The context rewriter is only assuming either BPF_LDX_MEM- or\n    BPF_STX_MEM-type operations, thus reject anything other than\n    that so that assumptions in the rewriter properly hold. Add\n    test cases as well for BPF selftests.\n\n    Fixes: d691f9e8d440 (\"bpf: allow programs to write to certain skb fields\")\n    Reported-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Signed-off-by: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Signed-off-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Signed-off-by: Greg Kroah-Hartman \u003cgregkh@linuxfoundation.org\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit fc8f593acb5fdfa686a5e4e3cd506ea23e81d1d2\nAuthor: Alexei Starovoitov \u003cast@kernel.org\u003e\nDate:   Mon Jan 29 02:49:00 2018 +0100\n\n    bpf: fix 32-bit divide by zero\n\n    [ upstream commit 68fda450a7df51cff9e5a4d4a4d9d0d5f2589153 ]\n\n    due to some JITs doing if (src_reg \u003d\u003d 0) check in 64-bit mode\n    for div/mod operations mask upper 32-bits of src register\n    before doing the check\n\n    Fixes: 622582786c9e (\"net: filter: x86: internal BPF JIT\")\n    Fixes: 7a12b5031c6b (\"sparc64: Add eBPF JIT.\")\n    Reported-by: syzbot+48340bb518e88849e2e3@syzkaller.appspotmail.com\n    Signed-off-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Signed-off-by: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Signed-off-by: Greg Kroah-Hartman \u003cgregkh@linuxfoundation.org\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 8379bad99cf4e0d63447c6c6b5cdc998328a7c15\nAuthor: Eric Dumazet \u003cedumazet@google.com\u003e\nDate:   Mon Jan 29 02:48:59 2018 +0100\n\n    bpf: fix divides by zero\n\n    [ upstream commit c366287ebd698ef5e3de300d90cd62ee9ee7373e ]\n\n    Divides by zero are not nice, lets avoid them if possible.\n\n    Also do_div() seems not needed when dealing with 32bit operands,\n    but this seems a minor detail.\n\n    Fixes: bd4cf0ed331a (\"net: filter: rework/optimize internal BPF interpreter\u0027s instruction set\")\n    Signed-off-by: Eric Dumazet \u003cedumazet@google.com\u003e\n    Reported-by: syzbot \u003csyzkaller@googlegroups.com\u003e\n    Signed-off-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Signed-off-by: Greg Kroah-Hartman \u003cgregkh@linuxfoundation.org\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 5469e46e82d4bec32a8752b8664569b571605ca3\nAuthor: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\nDate:   Mon Jan 29 02:48:57 2018 +0100\n\n    bpf: arsh is not supported in 32 bit alu thus reject it\n\n    [ upstream commit 7891a87efc7116590eaba57acc3c422487802c6f ]\n\n    The following snippet was throwing an \u0027unknown opcode cc\u0027 warning\n    in BPF interpreter:\n\n      0: (18) r0 \u003d 0x0\n      2: (7b) *(u64 *)(r10 -16) \u003d r0\n      3: (cc) (u32) r0 s\u003e\u003e\u003d (u32) r0\n      4: (95) exit\n\n    Although a number of JITs do support BPF_ALU | BPF_ARSH | BPF_{K,X}\n    generation, not all of them do and interpreter does neither. We can\n    leave existing ones and implement it later in bpf-next for the\n    remaining ones, but reject this properly in verifier for the time\n    being.\n\n    Fixes: 17a5267067f3 (\"bpf: verifier (add verifier core)\")\n    Reported-by: syzbot+93c4904c5c70348a6890@syzkaller.appspotmail.com\n    Signed-off-by: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Signed-off-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Signed-off-by: Greg Kroah-Hartman \u003cgregkh@linuxfoundation.org\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 6bfee5868d4e03e73d05d4024de45fc448d38411\nAuthor: Alexei Starovoitov \u003cast@kernel.org\u003e\nDate:   Mon Jan 29 02:48:56 2018 +0100\n\n    bpf: introduce BPF_JIT_ALWAYS_ON config\n\n    [ upstream commit 290af86629b25ffd1ed6232c4e9107da031705cb ]\n\n    The BPF interpreter has been used as part of the spectre 2 attack CVE-2017-5715.\n\n    A quote from goolge project zero blog:\n    \"At this point, it would normally be necessary to locate gadgets in\n    the host kernel code that can be used to actually leak data by reading\n    from an attacker-controlled location, shifting and masking the result\n    appropriately and then using the result of that as offset to an\n    attacker-controlled address for a load. But piecing gadgets together\n    and figuring out which ones work in a speculation context seems annoying.\n    So instead, we decided to use the eBPF interpreter, which is built into\n    the host kernel - while there is no legitimate way to invoke it from inside\n    a VM, the presence of the code in the host kernel\u0027s text section is sufficient\n    to make it usable for the attack, just like with ordinary ROP gadgets.\"\n\n    To make attacker job harder introduce BPF_JIT_ALWAYS_ON config\n    option that removes interpreter from the kernel in favor of JIT-only mode.\n    So far eBPF JIT is supported by:\n    x64, arm64, arm32, sparc64, s390, powerpc64, mips64\n\n    The start of JITed program is randomized and code page is marked as read-only.\n    In addition \"constant blinding\" can be turned on with net.core.bpf_jit_harden\n\n    v2-\u003ev3:\n    - move __bpf_prog_ret0 under ifdef (Daniel)\n\n    v1-\u003ev2:\n    - fix init order, test_bpf and cBPF (Daniel\u0027s feedback)\n    - fix offloaded bpf (Jakub\u0027s feedback)\n    - add \u0027return 0\u0027 dummy in case something can invoke prog-\u003ebpf_func\n    - retarget bpf tree. For bpf-next the patch would need one extra hunk.\n      It will be sent when the trees are merged back to net-next\n\n    Considered doing:\n      int bpf_jit_enable __read_mostly \u003d BPF_EBPF_JIT_DEFAULT;\n    but it seems better to land the patch as-is and in bpf-next remove\n    bpf_jit_enable global variable from all JITs, consolidate in one place\n    and remove this jit_init() function.\n\n    Signed-off-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Signed-off-by: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Signed-off-by: Greg Kroah-Hartman \u003cgregkh@linuxfoundation.org\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 4fcbe880a7f70d56508dc6d2c5748dcb6b6417ba\nAuthor: Alexei Starovoitov \u003cast@fb.com\u003e\nDate:   Mon Jan 29 02:48:55 2018 +0100\n\n    bpf: fix bpf_tail_call() x64 JIT\n\n    [ upstream commit 90caccdd8cc0215705f18b92771b449b01e2474a ]\n\n    - bpf prog_array just like all other types of bpf array accepts 32-bit index.\n      Clarify that in the comment.\n    - fix x64 JIT of bpf_tail_call which was incorrectly loading 8 instead of 4 bytes\n    - tighten corresponding check in the interpreter to stay consistent\n\n    The JIT bug can be triggered after introduction of BPF_F_NUMA_NODE flag\n    in commit 96eabe7a40aa in 4.14. Before that the map_flags would stay zero and\n    though JIT code is wrong it will check bounds correctly.\n    Hence two fixes tags. All other JITs don\u0027t have this problem.\n\n    Signed-off-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Fixes: 96eabe7a40aa (\"bpf: Allow selecting numa node during map creation\")\n    Fixes: b52f00e6a715 (\"x86: bpf_jit: implement bpf_tail_call() helper\")\n    Acked-by: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Acked-by: Martin KaFai Lau \u003ckafai@fb.com\u003e\n    Reviewed-by: Eric Dumazet \u003cedumazet@google.com\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Greg Kroah-Hartman \u003cgregkh@linuxfoundation.org\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit a3fd3f0275a74f5ff9bfc37b767fe5abc50b647f\nAuthor: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\nDate:   Wed Jan 10 23:25:05 2018 +0100\n\n    bpf, array: fix overflow in max_entries and undefined behavior in index_mask\n\n    commit bbeb6e4323dad9b5e0ee9f60c223dd532e2403b1 upstream.\n\n    syzkaller tried to alloc a map with 0xfffffffd entries out of a userns,\n    and thus unprivileged. With the recently added logic in b2157399cc98\n    (\"bpf: prevent out-of-bounds speculation\") we round this up to the next\n    power of two value for max_entries for unprivileged such that we can\n    apply proper masking into potentially zeroed out map slots.\n\n    However, this will generate an index_mask of 0xffffffff, and therefore\n    a + 1 will let this overflow into new max_entries of 0. This will pass\n    allocation, etc, and later on map access we still enforce on the original\n    attr-\u003emax_entries value which was 0xfffffffd, therefore triggering GPF\n    all over the place. Thus bail out on overflow in such case.\n\n    Moreover, on 32 bit archs roundup_pow_of_two() can also not be used,\n    since fls_long(max_entries - 1) can result in 32 and 1UL \u003c\u003c 32 in 32 bit\n    space is undefined. Therefore, do this by hand in a 64 bit variable.\n\n    This fixes all the issues triggered by syzkaller\u0027s reproducers.\n\n    Fixes: b2157399cc98 (\"bpf: prevent out-of-bounds speculation\")\n    Reported-by: syzbot+b0efb8e572d01bce1ae0@syzkaller.appspotmail.com\n    Reported-by: syzbot+6c15e9744f75f2364773@syzkaller.appspotmail.com\n    Reported-by: syzbot+d2f5524fb46fd3b312ee@syzkaller.appspotmail.com\n    Reported-by: syzbot+61d23c95395cc90dbc2b@syzkaller.appspotmail.com\n    Reported-by: syzbot+0d363c942452cca68c01@syzkaller.appspotmail.com\n    Signed-off-by: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Signed-off-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Signed-off-by: Greg Kroah-Hartman \u003cgregkh@linuxfoundation.org\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 46583a2655df94b4cf55bddaed7e7e1e3a9c9022\nAuthor: Alexei Starovoitov \u003cast@kernel.org\u003e\nDate:   Sun Jan 7 17:33:02 2018 -0800\n\n    bpf: prevent out-of-bounds speculation\n\n    commit b2157399cc9898260d6031c5bfe45fe137c1fbe7 upstream.\n\n    Under speculation, CPUs may mis-predict branches in bounds checks. Thus,\n    memory accesses under a bounds check may be speculated even if the\n    bounds check fails, providing a primitive for building a side channel.\n\n    To avoid leaking kernel data round up array-based maps and mask the index\n    after bounds check, so speculated load with out of bounds index will load\n    either valid value from the array or zero from the padded area.\n\n    Unconditionally mask index for all array types even when max_entries\n    are not rounded to power of 2 for root user.\n    When map is created by unpriv user generate a sequence of bpf insns\n    that includes AND operation to make sure that JITed code includes\n    the same \u0027index \u0026 index_mask\u0027 operation.\n\n    If prog_array map is created by unpriv user replace\n      bpf_tail_call(ctx, map, index);\n    with\n      if (index \u003e\u003d max_entries) {\n        index \u0026\u003d map-\u003eindex_mask;\n        bpf_tail_call(ctx, map, index);\n      }\n    (along with roundup to power 2) to prevent out-of-bounds speculation.\n    There is secondary redundant \u0027if (index \u003e\u003d max_entries)\u0027 in the interpreter\n    and in all JITs, but they can be optimized later if necessary.\n\n    Other array-like maps (cpumap, devmap, sockmap, perf_event_array, cgroup_array)\n    cannot be used by unpriv, so no changes there.\n\n    That fixes bpf side of \"Variant 1: bounds check bypass (CVE-2017-5753)\" on\n    all architectures with and without JIT.\n\n    v2-\u003ev3:\n    Daniel noticed that attack potentially can be crafted via syscall commands\n    without loading the program, so add masking to those paths as well.\n\n    Signed-off-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Acked-by: John Fastabend \u003cjohn.fastabend@gmail.com\u003e\n    Signed-off-by: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Cc: Jiri Slaby \u003cjslaby@suse.cz\u003e\n    [ Backported to 4.9 - gregkh ]\n    Signed-off-by: Greg Kroah-Hartman \u003cgregkh@linuxfoundation.org\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 2a398d547a76faffcb7666c38341f0c166a4f919\nAuthor: Alexei Starovoitov \u003cast@fb.com\u003e\nDate:   Wed Mar 15 18:26:40 2017 -0700\n\n    bpf: refactor fixup_bpf_calls()\n\n    commit 79741b3bdec01a8628368fbcfccc7d189ed606cb upstream.\n\n    reduce indent and make it iterate over instructions similar to\n    convert_ctx_accesses(). Also convert hard BUG_ON into soft verifier error.\n\n    Signed-off-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Acked-by: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Cc: Jiri Slaby \u003cjslaby@suse.cz\u003e\n    [Backported to 4.9.y - gregkh]\n    Signed-off-by: Greg Kroah-Hartman \u003cgregkh@linuxfoundation.org\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit ca289fbdf612b7419fef48957bb6a5bc6c94e937\nAuthor: Alexei Starovoitov \u003cast@fb.com\u003e\nDate:   Wed Mar 15 18:26:39 2017 -0700\n\n    bpf: move fixup_bpf_calls() function\n\n    commit e245c5c6a5656e4d61aa7bb08e9694fd6e5b2b9d upstream.\n\n    no functional change.\n    move fixup_bpf_calls() to verifier.c\n    it\u0027s being refactored in the next patch\n\n    Signed-off-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Acked-by: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Cc: Jiri Slaby \u003cjslaby@suse.cz\u003e\n    [backported to 4.9 - gregkh]\n    Signed-off-by: Greg Kroah-Hartman \u003cgregkh@linuxfoundation.org\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit e06abe187875e684438cdd66167cfc1d3cfae4f7\nAuthor: Ben Hutchings \u003cben@decadent.org.uk\u003e\nDate:   Sat Dec 23 02:26:17 2017 +0000\n\n    bpf/verifier: Fix states_equal() comparison of pointer and UNKNOWN\n\n    An UNKNOWN_VALUE is not supposed to be derived from a pointer, unless\n    pointer leaks are allowed.  Therefore, states_equal() must not treat\n    a state with a pointer in a register as \"equal\" to a state with an\n    UNKNOWN_VALUE in that register.\n\n    This was fixed differently upstream, but the code around here was\n    largely rewritten in 4.14 by commit f1174f77b50c \"bpf/verifier: rework\n    value tracking\".  The bug can be detected by the bpf/verifier sub-test\n    \"pointer/scalar confusion in state equality check (way 1)\".\n\n    Signed-off-by: Ben Hutchings \u003cben@decadent.org.uk\u003e\n    Cc: Edward Cree \u003cecree@solarflare.com\u003e\n    Cc: Jann Horn \u003cjannh@google.com\u003e\n    Cc: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Cc: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit e7f00ca3c36a454254a676147a0469ca33a58dd9\nAuthor: Jann Horn \u003cjannh@google.com\u003e\nDate:   Fri Dec 22 16:29:05 2017 +0100\n\n    bpf: fix incorrect sign extension in check_alu_op()\n\n    [ Upstream commit 95a762e2c8c942780948091f8f2a4f32fce1ac6f ]\n\n    Distinguish between\n    BPF_ALU64|BPF_MOV|BPF_K (load 32-bit immediate, sign-extended to 64-bit)\n    and BPF_ALU|BPF_MOV|BPF_K (load 32-bit immediate, zero-padded to 64-bit);\n    only perform sign extension in the first case.\n\n    Starting with v4.14, this is exploitable by unprivileged users as long as\n    the unprivileged_bpf_disabled sysctl isn\u0027t set.\n\n    Debian assigned CVE-2017-16995 for this issue.\n\n    v3:\n     - add CVE number (Ben Hutchings)\n\n    Fixes: 484611357c19 (\"bpf: allow access into map value arrays\")\n    Signed-off-by: Jann Horn \u003cjannh@google.com\u003e\n    Acked-by: Edward Cree \u003cecree@solarflare.com\u003e\n    Signed-off-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Signed-off-by: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Signed-off-by: Greg Kroah-Hartman \u003cgregkh@linuxfoundation.org\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit ad060c2d83c42fab8c3e8615f1a178c09dd28cb5\nAuthor: Jann Horn \u003cjannh@google.com\u003e\nDate:   Fri Dec 22 16:29:04 2017 +0100\n\n    bpf: reject out-of-bounds stack pointer calculation\n\n    Reject programs that compute wildly out-of-bounds stack pointers.\n    Otherwise, pointers can be computed with an offset that doesn\u0027t fit into an\n    `int`, causing security issues in the stack memory access check (as well as\n    signed integer overflow during offset addition).\n\n    This is a fix specifically for the v4.9 stable tree because the mainline\n    code looks very different at this point.\n\n    Fixes: 7bca0a9702edf (\"bpf: enhance verifier to understand stack pointer arithmetic\")\n    Signed-off-by: Jann Horn \u003cjannh@google.com\u003e\n    Acked-by: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Signed-off-by: Greg Kroah-Hartman \u003cgregkh@linuxfoundation.org\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 5fc649fb72af20b335ff0b51589edf45e25189da\nAuthor: Alexei Starovoitov \u003cast@fb.com\u003e\nDate:   Fri Dec 22 16:29:03 2017 +0100\n\n    bpf: fix branch pruning logic\n\n    [ Upstream commit c131187db2d3fa2f8bf32fdf4e9a4ef805168467 ]\n\n    when the verifier detects that register contains a runtime constant\n    and it\u0027s compared with another constant it will prune exploration\n    of the branch that is guaranteed not to be taken at runtime.\n    This is all correct, but malicious program may be constructed\n    in such a way that it always has a constant comparison and\n    the other branch is never taken under any conditions.\n    In this case such path through the program will not be explored\n    by the verifier. It won\u0027t be taken at run-time either, but since\n    all instructions are JITed the malicious program may cause JITs\n    to complain about using reserved fields, etc.\n    To fix the issue we have to track the instructions explored by\n    the verifier and sanitize instructions that are dead at run time\n    with NOPs. We cannot reject such dead code, since llvm generates\n    it for valid C code, since it doesn\u0027t do as much data flow\n    analysis as the verifier does.\n\n    Fixes: 17a5267067f3 (\"bpf: verifier (add verifier core)\")\n    Signed-off-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Acked-by: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Signed-off-by: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Signed-off-by: Greg Kroah-Hartman \u003cgregkh@linuxfoundation.org\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 1c4b04339ed65e9212af21d97ea8492959241bc6\nAuthor: Alexei Starovoitov \u003cast@fb.com\u003e\nDate:   Fri Dec 22 16:29:02 2017 +0100\n\n    bpf: adjust insn_aux_data when patching insns\n\n    [ Upstream commit 8041902dae5299c1f194ba42d14383f734631009 ]\n\n    convert_ctx_accesses() replaces single bpf instruction with a set of\n    instructions. Adjust corresponding insn_aux_data while patching.\n    It\u0027s needed to make sure subsequent \u0027for(all insn)\u0027 loops\n    have matching insn and insn_aux_data.\n\n    Signed-off-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Acked-by: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Greg Kroah-Hartman \u003cgregkh@linuxfoundation.org\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 4b581372fe405e6474a914502e9abf3a9f8295d4\nAuthor: Eric Dumazet \u003cedumazet@google.com\u003e\nDate:   Tue Nov 14 17:15:50 2017 -0800\n\n    bpf: fix lockdep splat\n\n    [ Upstream commit 89ad2fa3f043a1e8daae193bcb5fe34d5f8caf28 ]\n\n    pcpu_freelist_pop() needs the same lockdep awareness than\n    pcpu_freelist_populate() to avoid a false positive.\n\n     [ INFO: SOFTIRQ-safe -\u003e SOFTIRQ-unsafe lock order detected ]\n\n     switchto-defaul/12508 [HC0[0]:SC0[6]:HE0:SE0] is trying to acquire:\n      (\u0026htab-\u003ebuckets[i].lock){......}, at: [\u003cffffffff9dc099cb\u003e] __htab_percpu_map_update_elem+0x1cb/0x300\n\n     and this task is already holding:\n      (dev_queue-\u003edev-\u003eqdisc_class ?: \u0026qdisc_tx_lock#2){+.-...}, at: [\u003cffffffff9e135848\u003e] __dev_queue_xmit+0\n    x868/0x1240\n     which would create a new lock dependency:\n      (dev_queue-\u003edev-\u003eqdisc_class ?: \u0026qdisc_tx_lock#2){+.-...} -\u003e (\u0026htab-\u003ebuckets[i].lock){......}\n\n     but this new dependency connects a SOFTIRQ-irq-safe lock:\n      (dev_queue-\u003edev-\u003eqdisc_class ?: \u0026qdisc_tx_lock#2){+.-...}\n     ... which became SOFTIRQ-irq-safe at:\n       [\u003cffffffff9db5931b\u003e] __lock_acquire+0x42b/0x1f10\n       [\u003cffffffff9db5b32c\u003e] lock_acquire+0xbc/0x1b0\n       [\u003cffffffff9da05e38\u003e] _raw_spin_lock+0x38/0x50\n       [\u003cffffffff9e135848\u003e] __dev_queue_xmit+0x868/0x1240\n       [\u003cffffffff9e136240\u003e] dev_queue_xmit+0x10/0x20\n       [\u003cffffffff9e1965d9\u003e] ip_finish_output2+0x439/0x590\n       [\u003cffffffff9e197410\u003e] ip_finish_output+0x150/0x2f0\n       [\u003cffffffff9e19886d\u003e] ip_output+0x7d/0x260\n       [\u003cffffffff9e19789e\u003e] ip_local_out+0x5e/0xe0\n       [\u003cffffffff9e197b25\u003e] ip_queue_xmit+0x205/0x620\n       [\u003cffffffff9e1b8398\u003e] tcp_transmit_skb+0x5a8/0xcb0\n       [\u003cffffffff9e1ba152\u003e] tcp_write_xmit+0x242/0x1070\n       [\u003cffffffff9e1baffc\u003e] __tcp_push_pending_frames+0x3c/0xf0\n       [\u003cffffffff9e1b3472\u003e] tcp_rcv_established+0x312/0x700\n       [\u003cffffffff9e1c1acc\u003e] tcp_v4_do_rcv+0x11c/0x200\n       [\u003cffffffff9e1c3dc2\u003e] tcp_v4_rcv+0xaa2/0xc30\n       [\u003cffffffff9e191107\u003e] ip_local_deliver_finish+0xa7/0x240\n       [\u003cffffffff9e191a36\u003e] ip_local_deliver+0x66/0x200\n       [\u003cffffffff9e19137d\u003e] ip_rcv_finish+0xdd/0x560\n       [\u003cffffffff9e191e65\u003e] ip_rcv+0x295/0x510\n       [\u003cffffffff9e12ff88\u003e] __netif_receive_skb_core+0x988/0x1020\n       [\u003cffffffff9e130641\u003e] __netif_receive_skb+0x21/0x70\n       [\u003cffffffff9e1306ff\u003e] process_backlog+0x6f/0x230\n       [\u003cffffffff9e132129\u003e] net_rx_action+0x229/0x420\n       [\u003cffffffff9da07ee8\u003e] __do_softirq+0xd8/0x43d\n       [\u003cffffffff9e282bcc\u003e] do_softirq_own_stack+0x1c/0x30\n       [\u003cffffffff9dafc2f5\u003e] do_softirq+0x55/0x60\n       [\u003cffffffff9dafc3a8\u003e] __local_bh_enable_ip+0xa8/0xb0\n       [\u003cffffffff9db4c727\u003e] cpu_startup_entry+0x1c7/0x500\n       [\u003cffffffff9daab333\u003e] start_secondary+0x113/0x140\n\n     to a SOFTIRQ-irq-unsafe lock:\n      (\u0026head-\u003elock){+.+...}\n     ... which became SOFTIRQ-irq-unsafe at:\n     ...  [\u003cffffffff9db5971f\u003e] __lock_acquire+0x82f/0x1f10\n       [\u003cffffffff9db5b32c\u003e] lock_acquire+0xbc/0x1b0\n       [\u003cffffffff9da05e38\u003e] _raw_spin_lock+0x38/0x50\n       [\u003cffffffff9dc0b7fa\u003e] pcpu_freelist_pop+0x7a/0xb0\n       [\u003cffffffff9dc08b2c\u003e] htab_map_alloc+0x50c/0x5f0\n       [\u003cffffffff9dc00dc5\u003e] SyS_bpf+0x265/0x1200\n       [\u003cffffffff9e28195f\u003e] entry_SYSCALL_64_fastpath+0x12/0x17\n\n     other info that might help us debug this:\n\n     Chain exists of:\n       dev_queue-\u003edev-\u003eqdisc_class ?: \u0026qdisc_tx_lock#2 --\u003e \u0026htab-\u003ebuckets[i].lock --\u003e \u0026head-\u003elock\n\n      Possible interrupt unsafe locking scenario:\n\n            CPU0                    CPU1\n            ----                    ----\n       lock(\u0026head-\u003elock);\n                                    local_irq_disable();\n                                    lock(dev_queue-\u003edev-\u003eqdisc_class ?: \u0026qdisc_tx_lock#2);\n                                    lock(\u0026htab-\u003ebuckets[i].lock);\n       \u003cInterrupt\u003e\n         lock(dev_queue-\u003edev-\u003eqdisc_class ?: \u0026qdisc_tx_lock#2);\n\n      *** DEADLOCK ***\n\n    Fixes: e19494edab82 (\"bpf: introduce percpu_freelist\")\n    Signed-off-by: Eric Dumazet \u003cedumazet@google.com\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Sasha Levin \u003calexander.levin@verizon.com\u003e\n    Signed-off-by: Greg Kroah-Hartman \u003cgregkh@linuxfoundation.org\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 42cd8952e743e3b356722237b110ed39cd49acdf\nAuthor: Chenbo Feng \u003cfengc@google.com\u003e\nDate:   Wed Oct 18 13:00:26 2017 -0700\n\n    UPSTREAM: selinux: bpf: Add addtional check for bpf object file receive\n\n    Introduce a bpf object related check when sending and receiving files\n    through unix domain socket as well as binder. It checks if the receiving\n    process have privilege to read/write the bpf map or use the bpf program.\n    This check is necessary because the bpf maps and programs are using a\n    anonymous inode as their shared inode so the normal way of checking the\n    files and sockets when passing between processes cannot work properly on\n    eBPF object. This check only works when the BPF_SYSCALL is configured.\n\n    Signed-off-by: Chenbo Feng \u003cfengc@google.com\u003e\n    Acked-by: Stephen Smalley \u003csds@tycho.nsa.gov\u003e\n    Reviewed-by: James Morris \u003cjames.l.morris@oracle.com\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n\n    (cherry-pick from net-next: f66e448cfda021b0bcd884f26709796fe19c7cc1)\n    Bug: 30950746\n\n    Change-Id: I5b2cf4ccb4eab7eda91ddd7091d6aa3e7ed9f2cd\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit b760311edd0c8918efddc8161b4ba1127cc00cd8\nAuthor: Chenbo Feng \u003cfengc@google.com\u003e\nDate:   Wed Oct 18 13:00:25 2017 -0700\n\n    UPSTREAM: selinux: bpf: Add selinux check for eBPF syscall operations\n\n    Implement the actual checks introduced to eBPF related syscalls. This\n    implementation use the security field inside bpf object to store a sid that\n    identify the bpf object. And when processes try to access the object,\n    selinux will check if processes have the right privileges. The creation\n    of eBPF object are also checked at the general bpf check hook and new\n    cmd introduced to eBPF domain can also be checked there.\n\n    Signed-off-by: Chenbo Feng \u003cfengc@google.com\u003e\n    Acked-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Reviewed-by: James Morris \u003cjames.l.morris@oracle.com\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n\n    (cherry-pick from net-next: ec27c3568a34c7fe5fcf4ac0a354eda77687f7eb)\n    Bug: 30950746\n    Change-Id: Ifb0cdd4b7d470223b143646b339ba511ac77c156\n\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n    Change-Id: If49bc4c91c152efd36b372f8dfa15486e916f7df\n\ncommit 54b197e92fb0da064e7425a6646a52543e5071ee\nAuthor: Chenbo Feng \u003cfengc@google.com\u003e\nDate:   Wed Oct 18 13:00:24 2017 -0700\n\n    BACKPORT: security: bpf: Add LSM hooks for bpf object related syscall\n\n    Introduce several LSM hooks for the syscalls that will allow the\n    userspace to access to eBPF object such as eBPF programs and eBPF maps.\n    The security check is aimed to enforce a per object security protection\n    for eBPF object so only processes with the right priviliges can\n    read/write to a specific map or use a specific eBPF program. Besides\n    that, a general security hook is added before the multiplexer of bpf\n    syscall to check the cmd and the attribute used for the command. The\n    actual security module can decide which command need to be checked and\n    how the cmd should be checked.\n\n    Signed-off-by: Chenbo Feng \u003cfengc@google.com\u003e\n    Acked-by: James Morris \u003cjames.l.morris@oracle.com\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n\n    Added the LIST_HEAD_INIT call for security hooks, it nolonger exist in\n    uptream code.\n    (cherry-pick from net-next: afdb09c720b62b8090584c11151d856df330e57d)\n    Bug: 30950746\n\n    Change-Id: Ieb3ac74392f531735fc7c949b83346a5f587a77b\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 636d72b2d2ab8a78ada6f2ec48e9ecd83a3479ab\nAuthor: Chenbo Feng \u003cfengc@google.com\u003e\nDate:   Wed Oct 18 13:00:22 2017 -0700\n\n    BACKPORT: bpf: Add file mode configuration into bpf maps\n\n    Introduce the map read/write flags to the eBPF syscalls that returns the\n    map fd. The flags is used to set up the file mode when construct a new\n    file descriptor for bpf maps. To not break the backward capability, the\n    f_flags is set to O_RDWR if the flag passed by syscall is 0. Otherwise\n    it should be O_RDONLY or O_WRONLY. When the userspace want to modify or\n    read the map content, it will check the file mode to see if it is\n    allowed to make the change.\n\n    Signed-off-by: Chenbo Feng \u003cfengc@google.com\u003e\n    Acked-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Acked-by: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n\n    Deleted the file mode configuration code in unsupported map type and\n    removed the file mode check in non-existing helper functions.\n    (cherry-pick from net-next: 6e71b04a82248ccf13a94b85cbc674a9fefe53f5)\n    Bug: 30950746\n\n    Change-Id: Icfad20f1abb77f91068d244fb0d87fa40824dd1b\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit e9abb113a304794156cb9e7d96755578ef098f62\nAuthor: Anay Wadhera \u003cawadhera@berkeley.edu\u003e\nDate:   Fri May 21 13:29:12 2021 -0700\n\n    bpf: move bpf_map_show_fdinfo to match upstream location\n\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 68e67ead30cd7d656604eeea3d793c33d7f45f0f\nAuthor: Edward Cree \u003cecree@solarflare.com\u003e\nDate:   Fri Sep 15 14:37:38 2017 +0100\n\n    bpf/verifier: reject BPF_ALU64|BPF_END\n\n    [ Upstream commit e67b8a685c7c984e834e3181ef4619cd7025a136 ]\n\n    Neither ___bpf_prog_run nor the JITs accept it.\n    Also adds a new test case.\n\n    Fixes: 17a5267067f3 (\"bpf: verifier (add verifier core)\")\n    Signed-off-by: Edward Cree \u003cecree@solarflare.com\u003e\n    Acked-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Acked-by: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Greg Kroah-Hartman \u003cgregkh@linuxfoundation.org\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 10272c98adaafe9cfbac01e5e906e8b93eb41df2\nAuthor: Edward Cree \u003cecree@solarflare.com\u003e\nDate:   Fri Jul 21 14:37:34 2017 +0100\n\n    bpf/verifier: fix min/max handling in BPF_SUB\n\n    [ Upstream commit 9305706c2e808ae59f1eb201867f82f1ddf6d7a6 ]\n\n    We have to subtract the src max from the dst min, and vice-versa, since\n     (e.g.) the smallest result comes from the largest subtrahend.\n\n    Fixes: 484611357c19 (\"bpf: allow access into map value arrays\")\n    Signed-off-by: Edward Cree \u003cecree@solarflare.com\u003e\n    Acked-by: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Greg Kroah-Hartman \u003cgregkh@linuxfoundation.org\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit b56743acc363807bc6f9f4e54409debe32ff0d74\nAuthor: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\nDate:   Fri Jul 21 00:00:21 2017 +0200\n\n    bpf: fix mixed signed/unsigned derived min/max value bounds\n\n    [ Upstream commit 4cabc5b186b5427b9ee5a7495172542af105f02b ]\n\n    Edward reported that there\u0027s an issue in min/max value bounds\n    tracking when signed and unsigned compares both provide hints\n    on limits when having unknown variables. E.g. a program such\n    as the following should have been rejected:\n\n       0: (7a) *(u64 *)(r10 -8) \u003d 0\n       1: (bf) r2 \u003d r10\n       2: (07) r2 +\u003d -8\n       3: (18) r1 \u003d 0xffff8a94cda93400\n       5: (85) call bpf_map_lookup_elem#1\n       6: (15) if r0 \u003d\u003d 0x0 goto pc+7\n      R0\u003dmap_value(ks\u003d8,vs\u003d8,id\u003d0),min_value\u003d0,max_value\u003d0 R10\u003dfp\n       7: (7a) *(u64 *)(r10 -16) \u003d -8\n       8: (79) r1 \u003d *(u64 *)(r10 -16)\n       9: (b7) r2 \u003d -1\n      10: (2d) if r1 \u003e r2 goto pc+3\n      R0\u003dmap_value(ks\u003d8,vs\u003d8,id\u003d0),min_value\u003d0,max_value\u003d0 R1\u003dinv,min_value\u003d0\n      R2\u003dimm-1,max_value\u003d18446744073709551615,min_align\u003d1 R10\u003dfp\n      11: (65) if r1 s\u003e 0x1 goto pc+2\n      R0\u003dmap_value(ks\u003d8,vs\u003d8,id\u003d0),min_value\u003d0,max_value\u003d0 R1\u003dinv,min_value\u003d0,max_value\u003d1\n      R2\u003dimm-1,max_value\u003d18446744073709551615,min_align\u003d1 R10\u003dfp\n      12: (0f) r0 +\u003d r1\n      13: (72) *(u8 *)(r0 +0) \u003d 0\n      R0\u003dmap_value_adj(ks\u003d8,vs\u003d8,id\u003d0),min_value\u003d0,max_value\u003d1 R1\u003dinv,min_value\u003d0,max_value\u003d1\n      R2\u003dimm-1,max_value\u003d18446744073709551615,min_align\u003d1 R10\u003dfp\n      14: (b7) r0 \u003d 0\n      15: (95) exit\n\n    What happens is that in the first part ...\n\n       8: (79) r1 \u003d *(u64 *)(r10 -16)\n       9: (b7) r2 \u003d -1\n      10: (2d) if r1 \u003e r2 goto pc+3\n\n    ... r1 carries an unsigned value, and is compared as unsigned\n    against a register carrying an immediate. Verifier deduces in\n    reg_set_min_max() that since the compare is unsigned and operation\n    is greater than (\u003e), that in the fall-through/false case, r1\u0027s\n    minimum bound must be 0 and maximum bound must be r2. Latter is\n    larger than the bound and thus max value is reset back to being\n    \u0027invalid\u0027 aka BPF_REGISTER_MAX_RANGE. Thus, r1 state is now\n    \u0027R1\u003dinv,min_value\u003d0\u0027. The subsequent test ...\n\n      11: (65) if r1 s\u003e 0x1 goto pc+2\n\n    ... is a signed compare of r1 with immediate value 1. Here,\n    verifier deduces in reg_set_min_max() that since the compare\n    is signed this time and operation is greater than (\u003e), that\n    in the fall-through/false case, we can deduce that r1\u0027s maximum\n    bound must be 1, meaning with prior test, we result in r1 having\n    the following state: R1\u003dinv,min_value\u003d0,max_value\u003d1. Given that\n    the actual value this holds is -8, the bounds are wrongly deduced.\n    When this is being added to r0 which holds the map_value(_adj)\n    type, then subsequent store access in above case will go through\n    check_mem_access() which invokes check_map_access_adj(), that\n    will then probe whether the map memory is in bounds based\n    on the min_value and max_value as well as access size since\n    the actual unknown value is min_value \u003c\u003d x \u003c\u003d max_value; commit\n    fce366a9dd0d (\"bpf, verifier: fix alu ops against map_value{,\n    _adj} register types\") provides some more explanation on the\n    semantics.\n\n    It\u0027s worth to note in this context that in the current code,\n    min_value and max_value tracking are used for two things, i)\n    dynamic map value access via check_map_access_adj() and since\n    commit 06c1c049721a (\"bpf: allow helpers access to variable memory\")\n    ii) also enforced at check_helper_mem_access() when passing a\n    memory address (pointer to packet, map value, stack) and length\n    pair to a helper and the length in this case is an unknown value\n    defining an access range through min_value/max_value in that\n    case. The min_value/max_value tracking is /not/ used in the\n    direct packet access case to track ranges. However, the issue\n    also affects case ii), for example, the following crafted program\n    based on the same principle must be rejected as well:\n\n       0: (b7) r2 \u003d 0\n       1: (bf) r3 \u003d r10\n       2: (07) r3 +\u003d -512\n       3: (7a) *(u64 *)(r10 -16) \u003d -8\n       4: (79) r4 \u003d *(u64 *)(r10 -16)\n       5: (b7) r6 \u003d -1\n       6: (2d) if r4 \u003e r6 goto pc+5\n      R1\u003dctx R2\u003dimm0,min_value\u003d0,max_value\u003d0,min_align\u003d2147483648 R3\u003dfp-512\n      R4\u003dinv,min_value\u003d0 R6\u003dimm-1,max_value\u003d18446744073709551615,min_align\u003d1 R10\u003dfp\n       7: (65) if r4 s\u003e 0x1 goto pc+4\n      R1\u003dctx R2\u003dimm0,min_value\u003d0,max_value\u003d0,min_align\u003d2147483648 R3\u003dfp-512\n      R4\u003dinv,min_value\u003d0,max_value\u003d1 R6\u003dimm-1,max_value\u003d18446744073709551615,min_align\u003d1\n      R10\u003dfp\n       8: (07) r4 +\u003d 1\n       9: (b7) r5 \u003d 0\n      10: (6a) *(u16 *)(r10 -512) \u003d 0\n      11: (85) call bpf_skb_load_bytes#26\n      12: (b7) r0 \u003d 0\n      13: (95) exit\n\n    Meaning, while we initialize the max_value stack slot that the\n    verifier thinks we access in the [1,2] range, in reality we\n    pass -7 as length which is interpreted as u32 in the helper.\n    Thus, this issue is relevant also for the case of helper ranges.\n    Resetting both bounds in check_reg_overflow() in case only one\n    of them exceeds limits is also not enough as similar test can be\n    created that uses values which are within range, thus also here\n    learned min value in r1 is incorrect when mixed with later signed\n    test to create a range:\n\n       0: (7a) *(u64 *)(r10 -8) \u003d 0\n       1: (bf) r2 \u003d r10\n       2: (07) r2 +\u003d -8\n       3: (18) r1 \u003d 0xffff880ad081fa00\n       5: (85) call bpf_map_lookup_elem#1\n       6: (15) if r0 \u003d\u003d 0x0 goto pc+7\n      R0\u003dmap_value(ks\u003d8,vs\u003d8,id\u003d0),min_value\u003d0,max_value\u003d0 R10\u003dfp\n       7: (7a) *(u64 *)(r10 -16) \u003d -8\n       8: (79) r1 \u003d *(u64 *)(r10 -16)\n       9: (b7) r2 \u003d 2\n      10: (3d) if r2 \u003e\u003d r1 goto pc+3\n      R0\u003dmap_value(ks\u003d8,vs\u003d8,id\u003d0),min_value\u003d0,max_value\u003d0 R1\u003dinv,min_value\u003d3\n      R2\u003dimm2,min_value\u003d2,max_value\u003d2,min_align\u003d2 R10\u003dfp\n      11: (65) if r1 s\u003e 0x4 goto pc+2\n      R0\u003dmap_value(ks\u003d8,vs\u003d8,id\u003d0),min_value\u003d0,max_value\u003d0\n      R1\u003dinv,min_value\u003d3,max_value\u003d4 R2\u003dimm2,min_value\u003d2,max_value\u003d2,min_align\u003d2 R10\u003dfp\n      12: (0f) r0 +\u003d r1\n      13: (72) *(u8 *)(r0 +0) \u003d 0\n      R0\u003dmap_value_adj(ks\u003d8,vs\u003d8,id\u003d0),min_value\u003d3,max_value\u003d4\n      R1\u003dinv,min_value\u003d3,max_value\u003d4 R2\u003dimm2,min_value\u003d2,max_value\u003d2,min_align\u003d2 R10\u003dfp\n      14: (b7) r0 \u003d 0\n      15: (95) exit\n\n    This leaves us with two options for fixing this: i) to invalidate\n    all prior learned information once we switch signed context, ii)\n    to track min/max signed and unsigned boundaries separately as\n    done in [0]. (Given latter introduces major changes throughout\n    the whole verifier, it\u0027s rather net-next material, thus this\n    patch follows option i), meaning we can derive bounds either\n    from only signed tests or only unsigned tests.) There is still the\n    case of adjust_reg_min_max_vals(), where we adjust bounds on ALU\n    operations, meaning programs like the following where boundaries\n    on the reg get mixed in context later on when bounds are merged\n    on the dst reg must get rejected, too:\n\n       0: (7a) *(u64 *)(r10 -8) \u003d 0\n       1: (bf) r2 \u003d r10\n       2: (07) r2 +\u003d -8\n       3: (18) r1 \u003d 0xffff89b2bf87ce00\n       5: (85) call bpf_map_lookup_elem#1\n       6: (15) if r0 \u003d\u003d 0x0 goto pc+6\n      R0\u003dmap_value(ks\u003d8,vs\u003d8,id\u003d0),min_value\u003d0,max_value\u003d0 R10\u003dfp\n       7: (7a) *(u64 *)(r10 -16) \u003d -8\n       8: (79) r1 \u003d *(u64 *)(r10 -16)\n       9: (b7) r2 \u003d 2\n      10: (3d) if r2 \u003e\u003d r1 goto pc+2\n      R0\u003dmap_value(ks\u003d8,vs\u003d8,id\u003d0),min_value\u003d0,max_value\u003d0 R1\u003dinv,min_value\u003d3\n      R2\u003dimm2,min_value\u003d2,max_value\u003d2,min_align\u003d2 R10\u003dfp\n      11: (b7) r7 \u003d 1\n      12: (65) if r7 s\u003e 0x0 goto pc+2\n      R0\u003dmap_value(ks\u003d8,vs\u003d8,id\u003d0),min_value\u003d0,max_value\u003d0 R1\u003dinv,min_value\u003d3\n      R2\u003dimm2,min_value\u003d2,max_value\u003d2,min_align\u003d2 R7\u003dimm1,max_value\u003d0 R10\u003dfp\n      13: (b7) r0 \u003d 0\n      14: (95) exit\n\n      from 12 to 15: R0\u003dmap_value(ks\u003d8,vs\u003d8,id\u003d0),min_value\u003d0,max_value\u003d0\n      R1\u003dinv,min_value\u003d3 R2\u003dimm2,min_value\u003d2,max_value\u003d2,min_align\u003d2 R7\u003dimm1,min_value\u003d1 R10\u003dfp\n      15: (0f) r7 +\u003d r1\n      16: (65) if r7 s\u003e 0x4 goto pc+2\n      R0\u003dmap_value(ks\u003d8,vs\u003d8,id\u003d0),min_value\u003d0,max_value\u003d0 R1\u003dinv,min_value\u003d3\n      R2\u003dimm2,min_value\u003d2,max_value\u003d2,min_align\u003d2 R7\u003dinv,min_value\u003d4,max_value\u003d4 R10\u003dfp\n      17: (0f) r0 +\u003d r7\n      18: (72) *(u8 *)(r0 +0) \u003d 0\n      R0\u003dmap_value_adj(ks\u003d8,vs\u003d8,id\u003d0),min_value\u003d4,max_value\u003d4 R1\u003dinv,min_value\u003d3\n      R2\u003dimm2,min_value\u003d2,max_value\u003d2,min_align\u003d2 R7\u003dinv,min_value\u003d4,max_value\u003d4 R10\u003dfp\n      19: (b7) r0 \u003d 0\n      20: (95) exit\n\n    Meaning, in adjust_reg_min_max_vals() we must also reset range\n    values on the dst when src/dst registers have mixed signed/\n    unsigned derived min/max value bounds with one unbounded value\n    as otherwise they can be added together deducing false boundaries.\n    Once both boundaries are established from either ALU ops or\n    compare operations w/o mixing signed/unsigned insns, then they\n    can safely be added to other regs also having both boundaries\n    established. Adding regs with one unbounded side to a map value\n    where the bounded side has been learned w/o mixing ops is\n    possible, but the resulting map value won\u0027t recover from that,\n    meaning such op is considered invalid on the time of actual\n    access. Invalid bounds are set on the dst reg in case i) src reg,\n    or ii) in case dst reg already had them. The only way to recover\n    would be to perform i) ALU ops but only \u0027add\u0027 is allowed on map\n    value types or ii) comparisons, but these are disallowed on\n    pointers in case they span a range. This is fine as only BPF_JEQ\n    and BPF_JNE may be performed on PTR_TO_MAP_VALUE_OR_NULL registers\n    which potentially turn them into PTR_TO_MAP_VALUE type depending\n    on the branch, so only here min/max value cannot be invalidated\n    for them.\n\n    In terms of state pruning, value_from_signed is considered\n    as well in states_equal() when dealing with adjusted map values.\n    With regards to breaking existing programs, there is a small\n    risk, but use-cases are rather quite narrow where this could\n    occur and mixing compares probably unlikely.\n\n    Joint work with Josef and Edward.\n\n      [0] https://lists.iovisor.org/pipermail/iovisor-dev/2017-June/000822.html\n\n    Fixes: 484611357c19 (\"bpf: allow access into map value arrays\")\n    Reported-by: Edward Cree \u003cecree@solarflare.com\u003e\n    Signed-off-by: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Signed-off-by: Edward Cree \u003cecree@solarflare.com\u003e\n    Signed-off-by: Josef Bacik \u003cjbacik@fb.com\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Greg Kroah-Hartman \u003cgregkh@linuxfoundation.org\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 155147252d31bdd6cd2b88d534b6e09c6af18e62\nAuthor: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\nDate:   Fri Mar 31 02:24:02 2017 +0200\n\n    bpf, verifier: fix alu ops against map_value{, _adj} register types\n\n    [ Upstream commit fce366a9dd0ddc47e7ce05611c266e8574a45116 ]\n\n    While looking into map_value_adj, I noticed that alu operations\n    directly on the map_value() resp. map_value_adj() register (any\n    alu operation on a map_value() register will turn it into a\n    map_value_adj() typed register) are not sufficiently protected\n    against some of the operations. Two non-exhaustive examples are\n    provided that the verifier needs to reject:\n\n     i) BPF_AND on r0 (map_value_adj):\n\n      0: (bf) r2 \u003d r10\n      1: (07) r2 +\u003d -8\n      2: (7a) *(u64 *)(r2 +0) \u003d 0\n      3: (18) r1 \u003d 0xbf842a00\n      5: (85) call bpf_map_lookup_elem#1\n      6: (15) if r0 \u003d\u003d 0x0 goto pc+2\n       R0\u003dmap_value(ks\u003d8,vs\u003d48,id\u003d0),min_value\u003d0,max_value\u003d0 R10\u003dfp\n      7: (57) r0 \u0026\u003d 8\n      8: (7a) *(u64 *)(r0 +0) \u003d 22\n       R0\u003dmap_value_adj(ks\u003d8,vs\u003d48,id\u003d0),min_value\u003d0,max_value\u003d8 R10\u003dfp\n      9: (95) exit\n\n      from 6 to 9: R0\u003dinv,min_value\u003d0,max_value\u003d0 R10\u003dfp\n      9: (95) exit\n      processed 10 insns\n\n    ii) BPF_ADD in 32 bit mode on r0 (map_value_adj):\n\n      0: (bf) r2 \u003d r10\n      1: (07) r2 +\u003d -8\n      2: (7a) *(u64 *)(r2 +0) \u003d 0\n      3: (18) r1 \u003d 0xc24eee00\n      5: (85) call bpf_map_lookup_elem#1\n      6: (15) if r0 \u003d\u003d 0x0 goto pc+2\n       R0\u003dmap_value(ks\u003d8,vs\u003d48,id\u003d0),min_value\u003d0,max_value\u003d0 R10\u003dfp\n      7: (04) (u32) r0 +\u003d (u32) 0\n      8: (7a) *(u64 *)(r0 +0) \u003d 22\n       R0\u003dmap_value_adj(ks\u003d8,vs\u003d48,id\u003d0),min_value\u003d0,max_value\u003d0 R10\u003dfp\n      9: (95) exit\n\n      from 6 to 9: R0\u003dinv,min_value\u003d0,max_value\u003d0 R10\u003dfp\n      9: (95) exit\n      processed 10 insns\n\n    Issue is, while min_value / max_value boundaries for the access\n    are adjusted appropriately, we change the pointer value in a way\n    that cannot be sufficiently tracked anymore from its origin.\n    Operations like BPF_{AND,OR,DIV,MUL,etc} on a destination register\n    that is PTR_TO_MAP_VALUE{,_ADJ} was probably unintended, in fact,\n    all the test cases coming with 484611357c19 (\"bpf: allow access\n    into map value arrays\") perform BPF_ADD only on the destination\n    register that is PTR_TO_MAP_VALUE_ADJ.\n\n    Only for UNKNOWN_VALUE register types such operations make sense,\n    f.e. with unknown memory content fetched initially from a constant\n    offset from the map value memory into a register. That register is\n    then later tested against lower / upper bounds, so that the verifier\n    can then do the tracking of min_value / max_value, and properly\n    check once that UNKNOWN_VALUE register is added to the destination\n    register with type PTR_TO_MAP_VALUE{,_ADJ}. This is also what the\n    original use-case is solving. Note, tracking on what is being\n    added is done through adjust_reg_min_max_vals() and later access\n    to the map value enforced with these boundaries and the given offset\n    from the insn through check_map_access_adj().\n\n    Tests will fail for non-root environment due to prohibited pointer\n    arithmetic, in particular in check_alu_op(), we bail out on the\n    is_pointer_value() check on the dst_reg (which is false in root\n    case as we allow for pointer arithmetic via env-\u003eallow_ptr_leaks).\n\n    Similarly to PTR_TO_PACKET, one way to fix it is to restrict the\n    allowed operations on PTR_TO_MAP_VALUE{,_ADJ} registers to 64 bit\n    mode BPF_ADD. The test_verifier suite runs fine after the patch\n    and it also rejects mentioned test cases.\n\n    Fixes: 484611357c19 (\"bpf: allow access into map value arrays\")\n    Signed-off-by: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Reviewed-by: Josef Bacik \u003cjbacik@fb.com\u003e\n    Acked-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Greg Kroah-Hartman \u003cgregkh@linuxfoundation.org\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 4706daec7b6714a59f210cdfbc1c899e8d5c7307\nAuthor: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\nDate:   Thu May 18 03:00:06 2017 +0200\n\n    bpf: adjust verifier heuristics\n\n    [ Upstream commit 3c2ce60bdd3d57051bf85615deec04a694473840 ]\n\n    Current limits with regards to processing program paths do not\n    really reflect today\u0027s needs anymore due to programs becoming\n    more complex and verifier smarter, keeping track of more data\n    such as const ALU operations, alignment tracking, spilling of\n    PTR_TO_MAP_VALUE_ADJ registers, and other features allowing for\n    smarter matching of what LLVM generates.\n\n    This also comes with the side-effect that we result in fewer\n    opportunities to prune search states and thus often need to do\n    more work to prove safety than in the past due to different\n    register states and stack layout where we mismatch. Generally,\n    it\u0027s quite hard to determine what caused a sudden increase in\n    complexity, it could be caused by something as trivial as a\n    single branch somewhere at the beginning of the program where\n    LLVM assigned a stack slot that is marked differently throughout\n    other branches and thus causing a mismatch, where verifier\n    then needs to prove safety for the whole rest of the program.\n    Subsequently, programs with even less than half the insn size\n    limit can get rejected. We noticed that while some programs\n    load fine under pre 4.11, they get rejected due to hitting\n    limits on more recent kernels. We saw that in the vast majority\n    of cases (90+%) pruning failed due to register mismatches. In\n    case of stack mismatches, majority of cases failed due to\n    different stack slot types (invalid, spill, misc) rather than\n    differences in spilled registers.\n\n    This patch makes pruning more aggressive by also adding markers\n    that sit at conditional jumps as well. Currently, we only mark\n    jump targets for pruning. For example in direct packet access,\n    these are usually error paths where we bail out. We found that\n    adding these markers, it can reduce number of processed insns\n    by up to 30%. Another option is to ignore reg-\u003eid in probing\n    PTR_TO_MAP_VALUE_OR_NULL registers, which can help pruning\n    slightly as well by up to 7% observed complexity reduction as\n    stand-alone. Meaning, if a previous path with register type\n    PTR_TO_MAP_VALUE_OR_NULL for map X was found to be safe, then\n    in the current state a PTR_TO_MAP_VALUE_OR_NULL register for\n    the same map X must be safe as well. Last but not least the\n    patch also adds a scheduling point and bumps the current limit\n    for instructions to be processed to a more adequate value.\n\n    Signed-off-by: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Acked-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Greg Kroah-Hartman \u003cgregkh@linuxfoundation.org\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit ccbff7ed5a69d67b88cdd5802628cdaffcf201eb\nAuthor: John Fastabend \u003cjohn.fastabend@gmail.com\u003e\nDate:   Sun Jul 2 02:13:30 2017 +0200\n\n    bpf, verifier: add additional patterns to evaluate_reg_imm_alu\n\n    [ Upstream commit 43188702b3d98d2792969a3377a30957f05695e6 ]\n\n    Currently the verifier does not track imm across alu operations when\n    the source register is of unknown type. This adds additional pattern\n    matching to catch this and track imm. We\u0027ve seen LLVM generating this\n    pattern while working on cilium.\n\n    Signed-off-by: John Fastabend \u003cjohn.fastabend@gmail.com\u003e\n    Acked-by: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Acked-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Greg Kroah-Hartman \u003cgregkh@linuxfoundation.org\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit a8a14af3f51bd3aaedebb89dd8c5293c653f4a69\nAuthor: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\nDate:   Thu Jun 29 03:04:59 2017 +0200\n\n    bpf: prevent leaking pointer via xadd on unpriviledged\n\n    commit 6bdf6abc56b53103324dfd270a86580306e1a232 upstream.\n\n    Leaking kernel addresses on unpriviledged is generally disallowed,\n    for example, verifier rejects the following:\n\n      0: (b7) r0 \u003d 0\n      1: (18) r2 \u003d 0xffff897e82304400\n      3: (7b) *(u64 *)(r1 +48) \u003d r2\n      R2 leaks addr into ctx\n\n    Doing pointer arithmetic on them is also forbidden, so that they\n    don\u0027t turn into unknown value and then get leaked out. However,\n    there\u0027s xadd as a special case, where we don\u0027t check the src reg\n    for being a pointer register, e.g. the following will pass:\n\n      0: (b7) r0 \u003d 0\n      1: (7b) *(u64 *)(r1 +48) \u003d r0\n      2: (18) r2 \u003d 0xffff897e82304400 ; map\n      4: (db) lock *(u64 *)(r1 +48) +\u003d r2\n      5: (95) exit\n\n    We could store the pointer into skb-\u003ecb, loose the type context,\n    and then read it out from there again to leak it eventually out\n    of a map value. Or more easily in a different variant, too:\n\n       0: (bf) r6 \u003d r1\n       1: (7a) *(u64 *)(r10 -8) \u003d 0\n       2: (bf) r2 \u003d r10\n       3: (07) r2 +\u003d -8\n       4: (18) r1 \u003d 0x0\n       6: (85) call bpf_map_lookup_elem#1\n       7: (15) if r0 \u003d\u003d 0x0 goto pc+3\n       R0\u003dmap_value(ks\u003d8,vs\u003d8,id\u003d0),min_value\u003d0,max_value\u003d0 R6\u003dctx R10\u003dfp\n       8: (b7) r3 \u003d 0\n       9: (7b) *(u64 *)(r0 +0) \u003d r3\n      10: (db) lock *(u64 *)(r0 +0) +\u003d r6\n      11: (b7) r0 \u003d 0\n      12: (95) exit\n\n      from 7 to 11: R0\u003dinv,min_value\u003d0,max_value\u003d0 R6\u003dctx R10\u003dfp\n      11: (b7) r0 \u003d 0\n      12: (95) exit\n\n    Prevent this by checking xadd src reg for pointer types. Also\n    add a couple of test cases related to this.\n\n    Fixes: 1be7f75d1668 (\"bpf: enable non-root eBPF programs\")\n    Fixes: 17a5267067f3 (\"bpf: verifier (add verifier core)\")\n    Signed-off-by: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Acked-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Acked-by: Martin KaFai Lau \u003ckafai@fb.com\u003e\n    Acked-by: Edward Cree \u003cecree@solarflare.com\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Greg Kroah-Hartman \u003cgregkh@linuxfoundation.org\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit a2a655b8d49d19263b4779d41600bc4f470aa1be\nAuthor: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\nDate:   Wed Jan 18 15:14:17 2017 +0100\n\n    bpf: don\u0027t trigger OOM killer under pressure with map alloc\n\n    [ Upstream commit d407bd25a204bd66b7346dde24bd3d37ef0e0b05 ]\n\n    This patch adds two helpers, bpf_map_area_alloc() and bpf_map_area_free(),\n    that are to be used for map allocations. Using kmalloc() for very large\n    allocations can cause excessive work within the page allocator, so i) fall\n    back earlier to vmalloc() when the attempt is considered costly anyway,\n    and even more importantly ii) don\u0027t trigger OOM killer with any of the\n    allocators.\n\n    Since this is based on a user space request, for example, when creating\n    maps with element pre-allocation, we really want such requests to fail\n    instead of killing other user space processes.\n\n    Also, don\u0027t spam the kernel log with warnings should any of the allocations\n    fail under pressure. Given that, we can make backend selection in\n    bpf_map_area_alloc() generic, and convert all maps over to use this API\n    for spots with potentially large allocation requests.\n\n    Note, replacing the one kmalloc_array() is fine as overflow checks happen\n    earlier in htab_map_alloc(), since it must also protect the multiplication\n    for vmalloc() should kmalloc_array() fail.\n\n    Signed-off-by: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Acked-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Sasha Levin \u003calexander.levin@verizon.com\u003e\n    Signed-off-by: Greg Kroah-Hartman \u003cgregkh@linuxfoundation.org\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit bfda5f80c50db00b4d2fe4e236e8381768599768\nAuthor: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\nDate:   Tue Jun 6 18:38:04 2017 +0200\n\n    FROMLIST: bpf: cgroup skb progs cannot access ld_abs/ind\n\n    Commit fb9a307d11d6 (\"bpf: Allow CGROUP_SKB eBPF program to\n    access sk_buff\") enabled programs of BPF_PROG_TYPE_CGROUP_SKB\n    type to use ld_abs/ind instructions. However, at this point,\n    we cannot use them, since offsets relative to SKF_LL_OFF will\n    end up pointing skb_mac_header(skb) out of bounds since in the\n    egress path it is not yet set at that point in time, but only\n    after __dev_queue_xmit() did a general reset on the mac header.\n    bpf_internal_load_pointer_neg_helper() will then end up reading\n    data from a wrong offset.\n\n    BPF_PROG_TYPE_CGROUP_SKB programs can use bpf_skb_load_bytes()\n    already to access packet data, which is also more flexible than\n    the insns carried over from cBPF.\n\n    Fixes: fb9a307d11d6 (\"bpf: Allow CGROUP_SKB eBPF program to access sk_buff\")\n    Signed-off-by: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Acked-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Cc: Chenbo Feng \u003cfengc@google.com\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n\n    (url: http://patchwork.ozlabs.org/patch/771946/)\n\n    Signed-off-by: Chenbo Feng \u003cfengc@google.com\u003e\n    Bug: 30950746\n    Change-Id: Ia32ac79d8c0d18f811ec101897284a8b60cb042a\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit f7021caa40d6683b832385c7de022f6742124749\nAuthor: Chenbo Feng \u003cfengc@google.com\u003e\nDate:   Fri Jun 2 17:24:31 2017 -0700\n\n    FROMLIST: [net-next,v2,2/2] bpf: Remove the capability check for cgroup skb eBPF program\n\n    Currently loading a cgroup skb eBPF program require a CAP_SYS_ADMIN\n    capability while attaching the program to a cgroup only requires the\n    user have CAP_NET_ADMIN privilege. We can escape the capability\n    check when load the program just like socket filter program to make\n    the capability requirement consistent.\n\n    Change since v1:\n    Change the code style in order to be compliant with checkpatch.pl\n    preference\n\n    (url: http://patchwork.ozlabs.org/patch/769460/)\n    Signed-off-by: Chenbo Feng \u003cfengc@google.com\u003e\n    Bug: 30950746\n    Change-Id: Ibe51235127d6f9349b8f563ad31effc061b278ed\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit d317d86203c7bd0023125fedf7a695e0713c7a69\nAuthor: Chenbo Feng \u003cfengc@google.com\u003e\nDate:   Fri Jun 2 17:04:59 2017 -0700\n\n    FROMLIST: [net-next,v2,1/2] bpf: Allow CGROUP_SKB eBPF program to access sk_buff\n\n    This allows cgroup eBPF program to classify packet based on their\n    protocol or other detail information. Currently program need\n    CAP_NET_ADMIN privilege to attach a cgroup eBPF program, and A\n    process with CAP_NET_ADMIN can already see all packets on the system,\n    for example, by creating an iptables rules that causes the packet to\n    be passed to userspace via NFLOG.\n\n    (url: http://patchwork.ozlabs.org/patch/769459/)\n\n    Signed-off-by: Chenbo Feng \u003cfengc@google.com\u003e\n    Bug: 30950746\n    Change-Id: I11bef84ce26cf8b8f1b89483c32a7fcdd61ae926\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 5c5b00bb89839f634e7ea6561c04376458567709\nAuthor: Daniel Mack \u003cdaniel@zonque.org\u003e\nDate:   Mon Nov 28 14:11:04 2016 +0100\n\n    UPSTREAM: bpf: cgroup: fix documentation of __cgroup_bpf_update()\n\n    There\u0027s a \u0027not\u0027 missing in one paragraph. Add it.\n\n    Fixes: 3007098494be (\"cgroup: add support for eBPF programs\")\n    Signed-off-by: Daniel Mack \u003cdaniel@zonque.org\u003e\n    Reported-by: Rami Rosen \u003croszenrami@gmail.com\u003e\n    Acked-by: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Acked-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n\n    Fixes: Change-Id: I3df35d8d3b1261503f9b5bcd90b18c9358f1ac28\n           (\"cgroup: add support for eBPF programs\")\n    (cherry picked from commit 01ae87eab53675cbdabd5c4d727c4a35e397cce0)\n    Signed-off-by: Amit Pundir \u003camit.pundir@linaro.org\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 8f630d15e6f6bbf94358b332177fcda77ba85c7f\nAuthor: Alexei Starovoitov \u003cast@fb.com\u003e\nDate:   Fri Feb 10 20:28:24 2017 -0800\n\n    BACKPORT: bpf: introduce BPF_F_ALLOW_OVERRIDE flag\n\n    If BPF_F_ALLOW_OVERRIDE flag is used in BPF_PROG_ATTACH command\n    to the given cgroup the descendent cgroup will be able to override\n    effective bpf program that was inherited from this cgroup.\n    By default it\u0027s not passed, therefore override is disallowed.\n\n    Examples:\n    1.\n    prog X attached to /A with default\n    prog Y fails to attach to /A/B and /A/B/C\n    Everything under /A runs prog X\n\n    2.\n    prog X attached to /A with allow_override.\n    prog Y fails to attach to /A/B with default (non-override)\n    prog M attached to /A/B with allow_override.\n    Everything under /A/B runs prog M only.\n\n    3.\n    prog X attached to /A with allow_override.\n    prog Y fails to attach to /A with default.\n    The user has to detach first to switch the mode.\n\n    In the future this behavior may be extended with a chain of\n    non-overridable programs.\n\n    Also fix the bug where detach from cgroup where nothing is attached\n    was not throwing error. Return ENOENT in such case.\n\n    Add several testcases and adjust libbpf.\n\n    Fixes: 3007098494be (\"cgroup: add support for eBPF programs\")\n    Signed-off-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Acked-by: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Acked-by: Tejun Heo \u003ctj@kernel.org\u003e\n    Acked-by: Daniel Mack \u003cdaniel@zonque.org\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n\n    Fixes: Change-Id: I3df35d8d3b1261503f9b5bcd90b18c9358f1ac28\n           (\"cgroup: add support for eBPF programs\")\n    [AmitP: Refactored original patch for android-4.9 where libbpf sources\n            are in samples/bpf/ and test_cgrp2_attach2, test_cgrp2_sock,\n            and test_cgrp2_sock2 sample tests do not exist.]\n    (cherry picked from commit 7f677633379b4abb3281cdbe7e7006f049305c03)\n    Signed-off-by: Amit Pundir \u003camit.pundir@linaro.org\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 0e3a272e683c012bfa1367a4faa8a6fecdce731a\nAuthor: Daniel Mack \u003cdaniel@zonque.org\u003e\nDate:   Wed Nov 23 16:52:30 2016 +0100\n\n    UPSTREAM: samples: bpf: add userspace example for attaching eBPF programs to cgroups\n\n    Cherry-pick from commit d8c5b17f2bc0de09fbbfa14d90e8168163a579e7\n\n    Add a simple userpace program to demonstrate the new API to attach eBPF\n    programs to cgroups. This is what it does:\n\n     * Create arraymap in kernel with 4 byte keys and 8 byte values\n\n     * Load eBPF program\n\n       The eBPF program accesses the map passed in to store two pieces of\n       information. The number of invocations of the program, which maps\n       to the number of packets received, is stored to key 0. Key 1 is\n       incremented on each iteration by the number of bytes stored in\n       the skb.\n\n     * Detach any eBPF program previously attached to the cgroup\n\n     * Attach the new program to the cgroup using BPF_PROG_ATTACH\n\n     * Once a second, read map[0] and map[1] to see how many bytes and\n       packets were seen on any socket of tasks in the given cgroup.\n\n    The program takes a cgroup path as 1st argument, and either \"ingress\"\n    or \"egress\" as 2nd. Optionally, \"drop\" can be passed as 3rd argument,\n    which will make the generated eBPF program return 0 instead of 1, so\n    the kernel will drop the packet.\n\n    libbpf gained two new wrappers for the new syscall commands.\n\n    Signed-off-by: Daniel Mack \u003cdaniel@zonque.org\u003e\n    Acked-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Bug: 30950746\n    Change-Id: I011436a755abd62050edd22e47995c166a0bd8a2\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit bee81948a552ad018cacc5661f10eac0c6360088\nAuthor: Daniel Mack \u003cdaniel@zonque.org\u003e\nDate:   Wed Nov 23 16:52:25 2016 +0100\n\n    UPSTREAM: bpf: add new prog type for cgroup socket filtering\n\n    Cherry-pick from commit 0e33661de493db325435d565a4a722120ae4cbf3\n\n    This program type is similar to BPF_PROG_TYPE_SOCKET_FILTER, except that\n    it does not allow BPF_LD_[ABS|IND] instructions and hooks up the\n    bpf_skb_load_bytes() helper.\n\n    Programs of this type will be attached to cgroups for network filtering\n    and accounting.\n\n    Signed-off-by: Daniel Mack \u003cdaniel@zonque.org\u003e\n    Acked-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Bug: 30950746\n    Change-Id: I7b9e063d5d7a91da80917c6d353a60b877133752\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit a17f4d37e594e967e30573d58bc07141b3d72dde\nAuthor: Willem de Bruijn \u003cwillemb@google.com\u003e\nDate:   Tue Apr 11 14:08:08 2017 -0400\n\n    BACKPORT: UPSTREAM: bpf: pass sk to helper functions\n\n    Cherrypick from commit 8f917bba0042f1e3b7693743fbe9782709e936e7\n\n    BPF helper functions access socket fields through skb-\u003esk. This is not\n    set in ingress cgroup and socket filters. The association is only made\n    in skb_set_owner_r once the filter has accepted the packet. Sk is\n    available as socket lookup has taken place.\n\n    Temporarily set skb-\u003esk to sk in these cases.\n\n    Signed-off-by: Willem de Bruijn \u003cwillemb@google.com\u003e\n    Acked-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Bug: 30950746\n    Change-Id: Ifcbcbe2ab2882dc79c56f9707be1d6aef08c7fd3\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit f8987f871e8401a504d87ef5ef5040cf1a02d16a\nAuthor: Daniel Mack \u003cdaniel@zonque.org\u003e\nDate:   Wed Nov 23 16:52:27 2016 +0100\n\n    UPSTREAM: bpf: add BPF_PROG_ATTACH and BPF_PROG_DETACH commands\n\n    Cherry-pick from commit f4324551489e8781d838f941b7aee4208e52e8bf\n\n    Extend the bpf(2) syscall by two new commands, BPF_PROG_ATTACH and\n    BPF_PROG_DETACH which allow attaching and detaching eBPF programs\n    to a target.\n\n    On the API level, the target could be anything that has an fd in\n    userspace, hence the name of the field in union bpf_attr is called\n    \u0027target_fd\u0027.\n\n    When called with BPF_ATTACH_TYPE_CGROUP_INET_{E,IN}GRESS, the target is\n    expected to be a valid file descriptor of a cgroup v2 directory which\n    has the bpf controller enabled. These are the only use-cases\n    implemented by this patch at this point, but more can be added.\n\n    If a program of the given type already exists in the given cgroup,\n    the program is swapped automically, so userspace does not have to drop\n    an existing program first before installing a new one, which would\n    otherwise leave a gap in which no program is attached.\n\n    For more information on the propagation logic to subcgroups, please\n    refer to the bpf cgroup controller implementation.\n\n    The API is guarded by CAP_NET_ADMIN.\n\n    Signed-off-by: Daniel Mack \u003cdaniel@zonque.org\u003e\n    Acked-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Bug: 30950746\n    Change-Id: Iab156859332166835d51e1e6f64e5cb8b81870f2\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 1b3d5978e1866f150ccad2ff421faab89589e5f5\nAuthor: Tejun Heo \u003ctj@kernel.org\u003e\nDate:   Fri Nov 20 15:55:52 2015 -0500\n\n    kernfs: implement kernfs_walk_and_get()\n\n    Implement kernfs_walk_and_get() which is similar to\n    kernfs_find_and_get() but can walk a path instead of just a name.\n\n    v2: Use strlcpy() instead of strlen() + memcpy() as suggested by\n        David.\n\n    Signed-off-by: Tejun Heo \u003ctj@kernel.org\u003e\n    Acked-by: Greg Kroah-Hartman \u003cgregkh@linuxfoundation.org\u003e\n    Cc: David Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 44f016625b070038fb36eee2bc346238683a1022\nAuthor: Johannes Weiner \u003channes@cmpxchg.org\u003e\nDate:   Thu Jan 31 10:41:54 2019 +1100\n\n    BACKPORT: fs: kernfs: add poll file operation\n\n    Patch series \"psi: pressure stall monitors\", v3.\n\n    Android is adopting psi to detect and remedy memory pressure that results\n    in stuttering and decreased responsiveness on mobile devices.\n\n    Psi gives us the stall information, but because we\u0027re dealing with\n    latencies in the millisecond range, periodically reading the pressure\n    files to detect stalls in a timely fashion is not feasible.  Psi also\n    doesn\u0027t aggregate its averages at a high enough frequency right now.\n\n    This patch series extends the psi interface such that users can configure\n    sensitive latency thresholds and use poll() and friends to be notified\n    when these are breached.\n\n    As high-frequency aggregation is costly, it implements an aggregation\n    method that is optimized for fast, short-interval averaging, and makes the\n    aggregation frequency adaptive, such that high-frequency updates only\n    happen while monitored stall events are actively occurring.\n\n    With these patches applied, Android can monitor for, and ward off,\n    mounting memory shortages before they cause problems for the user.  For\n    example, using memory stall monitors in userspace low memory killer daemon\n    (lmkd) we can detect mounting pressure and kill less important processes\n    before device becomes visibly sluggish.  In our memory stress testing psi\n    memory monitors produce roughly 10x less false positives compared to\n    vmpressure signals.  Having ability to specify multiple triggers for the\n    same psi metric allows other parts of Android framework to monitor memory\n    state of the device and act accordingly.\n\n    The new interface is straightforward.  The user opens one of the pressure\n    files for writing and writes a trigger description into the file\n    descriptor that defines the stall state - some or full, and the maximum\n    stall time over a given window of time.  E.g.:\n\n            /* Signal when stall time exceeds 100ms of a 1s window */\n            char trigger[] \u003d \"full 100000 1000000\";\n            fd \u003d open(\"/proc/pressure/memory\");\n            write(fd, trigger, sizeof(trigger));\n            while (poll() \u003e\u003d 0) {\n                    ...\n            }\n            close(fd);\n\n    When the monitored stall state is entered, psi adapts its aggregation\n    frequency according to what the configured time window requires in order\n    to emit event signals in a timely fashion.  Once the stalling subsides,\n    aggregation reverts back to normal.\n\n    The trigger is associated with the open file descriptor.  To stop\n    monitoring, the user only needs to close the file descriptor and the\n    trigger is discarded.\n\n    Patches 1-4 prepare the psi code for polling support.  Patch 5 implements\n    the adaptive polling logic, the pressure growth detection optimized for\n    short intervals, and hooks up write() and poll() on the pressure files.\n\n    The patches were developed in collaboration with Johannes Weiner.\n\n    This patch (of 5):\n\n    Kernfs has a standardized poll/notification mechanism for waking all\n    pollers on all fds when a filesystem node changes.  To allow polling for\n    custom events, add a .poll callback that can override the default.\n\n    This is in preparation for pollable cgroup pressure files which have\n    per-fd trigger configurations.\n\n    Link: http://lkml.kernel.org/r/20190124211518.244221-2-surenb@google.com\n    Signed-off-by: Johannes Weiner \u003channes@cmpxchg.org\u003e\n    Signed-off-by: Suren Baghdasaryan \u003csurenb@google.com\u003e\n    Cc: Dennis Zhou \u003cdennis@kernel.org\u003e\n    Cc: Ingo Molnar \u003cmingo@redhat.com\u003e\n    Cc: Jens Axboe \u003caxboe@kernel.dk\u003e\n    Cc: Li Zefan \u003clizefan@huawei.com\u003e\n    Cc: Peter Zijlstra \u003cpeterz@infradead.org\u003e\n    Cc: Tejun Heo \u003ctj@kernel.org\u003e\n    Signed-off-by: Andrew Morton \u003cakpm@linux-foundation.org\u003e\n    Signed-off-by: Stephen Rothwell \u003csfr@canb.auug.org.au\u003e\n\n    (cherry picked from commit: 147e1a97c4a0bdd43f55a582a9416bb9092563a9)\n\n    Conflicts:\n            fs/kernfs/file.c\n            include/linux/kernfs.h\n\n    1. replaced __poll_t with unsigned int.\n    2. replaced kernfs_dentry_node() with dentry-\u003ed_fsdata\n    3. replaced EPOLLERR/EPOLLPRI with POLLERR/POLLPRI (values are the same)\n\n    Bug: 127712811\n    Test: lmkd in PSI mode\n    Change-Id: Ic2bed334d05aec62f4e695f263893c3057921c55\n    Signed-off-by: Suren Baghdasaryan \u003csurenb@google.com\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit f86fda4c2673b0798c59ecf4490d5710f76bd3a0\nAuthor: Tejun Heo \u003ctj@kernel.org\u003e\nDate:   Tue Dec 27 14:49:03 2016 -0500\n\n    UPSTREAM: kernfs: add kernfs_ops-\u003eopen/release() callbacks\n\n    Add -\u003eopen/release() methods to kernfs_ops.  -\u003eopen() is called when\n    the file is opened and -\u003erelease() when the file is either released or\n    severed.  These callbacks can be used, for example, to manage\n    persistent caching objects over multiple seq_file iterations.\n\n    Signed-off-by: Tejun Heo \u003ctj@kernel.org\u003e\n    Acked-by: Greg Kroah-Hartman \u003cgregkh@linuxfoundation.org\u003e\n    Acked-by: Acked-by: Zefan Li \u003clizefan@huawei.com\u003e\n\n    (cherry picked from commit 0e67db2f9fe91937e798e3d7d22c50a8438187e1)\n\n    Bug: 111308141\n    Test: modified lmkd to use PSI and tested using lmkd_unit_test\n\n    Change-Id: Id06e9d5c6da1280bcdd4dc86309dcfaf52b8f9a4\n    Signed-off-by: Suren Baghdasaryan \u003csurenb@google.com\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 4ce51c00483e190498157a34dde66876d5aeb395\nAuthor: Aditya Kali \u003cadityakali@google.com\u003e\nDate:   Fri Jan 29 02:54:04 2016 -0600\n\n    kernfs: Add API to generate relative kernfs path\n\n    The new function kernfs_path_from_node() generates and returns kernfs\n    path of a given kernfs_node relative to a given parent kernfs_node.\n\n    Signed-off-by: Aditya Kali \u003cadityakali@google.com\u003e\n    Signed-off-by: Serge E. Hallyn \u003cserge.hallyn@canonical.com\u003e\n    Acked-by: Greg Kroah-Hartman \u003cgregkh@linuxfoundation.org\u003e\n    Signed-off-by: Tejun Heo \u003ctj@kernel.org\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit be868eec9d8f486310ebaecfd42e88089c10bf77\nAuthor: Aditya Kali \u003cadityakali@google.com\u003e\nDate:   Fri Jan 29 02:54:05 2016 -0600\n\n    sched: new clone flag CLONE_NEWCGROUP for cgroup namespace\n\n    CLONE_NEWCGROUP will be used to create new cgroup namespace.\n\n    Signed-off-by: Aditya Kali \u003cadityakali@google.com\u003e\n    Signed-off-by: Serge Hallyn \u003cserge.hallyn@canonical.com\u003e\n    Signed-off-by: Tejun Heo \u003ctj@kernel.org\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 566a2b0fcc5b66893329385bf9207d18ed5a3e97\nAuthor: Daniel Mack \u003cdaniel@zonque.org\u003e\nDate:   Wed Nov 23 16:52:26 2016 +0100\n\n    UPSTREAM: cgroup: add support for eBPF programs\n\n    Cherry-pick from commit 3007098494bec614fb55dee7bc0410bb7db5ad18\n\n    This patch adds two sets of eBPF program pointers to struct cgroup.\n    One for such that are directly pinned to a cgroup, and one for such\n    that are effective for it.\n\n    To illustrate the logic behind that, assume the following example\n    cgroup hierarchy.\n\n      A - B - C\n            \\ D - E\n\n    If only B has a program attached, it will be effective for B, C, D\n    and E. If D then attaches a program itself, that will be effective for\n    both D and E, and the program in B will only affect B and C. Only one\n    program of a given type is effective for a cgroup.\n\n    Attaching and detaching programs will be done through the bpf(2)\n    syscall. For now, ingress and egress inet socket filtering are the\n    only supported use-cases.\n\n    Signed-off-by: Daniel Mack \u003cdaniel@zonque.org\u003e\n    Acked-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Bug: 30950746\n    Change-Id: I3df35d8d3b1261503f9b5bcd90b18c9358f1ac28\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 0216c431c114d91feb93ebda7b55748f4002754c\nAuthor: Tejun Heo \u003ctj@kernel.org\u003e\nDate:   Thu Jan 26 16:47:28 2017 -0500\n\n    cgroup: don\u0027t online subsystems before cgroup_name/path() are operational\n\n    commit 07cd12945551b63ecb1a349d50a6d69d1d6feb4a upstream.\n\n    While refactoring cgroup creation, a5bca2152036 (\"cgroup: factor out\n    cgroup_create() out of cgroup_mkdir()\") incorrectly onlined subsystems\n    before the new cgroup is associated with it kernfs_node.  This is fine\n    for cgroup proper but cgroup_name/path() depend on the associated\n    kernfs_node and if a subsystem makes the new cgroup_subsys_state\n    visible, which they\u0027re allowed to after onlining, it can lead to NULL\n    dereference.\n\n    The current code performs cgroup creation and subsystem onlining in\n    cgroup_create() and cgroup_mkdir() makes the cgroup and subsystems\n    visible afterwards.  There\u0027s no reason to online the subsystems early\n    and we can simply drop cgroup_apply_control_enable() call from\n    cgroup_create() so that the subsystems are onlined and made visible at\n    the same time.\n\n    Signed-off-by: Tejun Heo \u003ctj@kernel.org\u003e\n    Reported-by: Konstantin Khlebnikov \u003ckhlebnikov@yandex-team.ru\u003e\n    Fixes: a5bca2152036 (\"cgroup: factor out cgroup_create() out of cgroup_mkdir()\")\n    Signed-off-by: Greg Kroah-Hartman \u003cgregkh@linuxfoundation.org\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 6246469ce54c6317ea480fd4263753fad8b078f6\nAuthor: Tejun Heo \u003ctj@kernel.org\u003e\nDate:   Thu Sep 29 15:49:40 2016 +0200\n\n    cgroup: fix error handling regressions in proc_cgroup_show() and cgroup_release_agent()\n\n    4c737b41de7f (\"cgroup: make cgroup_path() and friends behave in the\n    style of strlcpy()\") broke error handling in proc_cgroup_show() and\n    cgroup_release_agent() by not handling negative return values from\n    cgroup_path_ns_locked().  Fix it.\n\n    Reported-by: Dan Carpenter \u003cdan.carpenter@oracle.com\u003e\n    Signed-off-by: Tejun Heo \u003ctj@kernel.org\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n    Change-Id: If1dbcdb90f9eefbfc2aa245a8fc4da5b15e23296\n\ncommit 0ccc12c3c66392dfe9c14caeb8a3f8f35a8793ba\nAuthor: Tejun Heo \u003ctj@kernel.org\u003e\nDate:   Fri Sep 23 16:55:49 2016 -0400\n\n    cgroup: fix invalid controller enable rejections with cgroup namespace\n\n    On the v2 hierarchy, \"cgroup.subtree_control\" rejects controller\n    enables if the cgroup has processes in it.  The enforcement of this\n    logic assumes that the cgroup wouldn\u0027t have any css_sets associated\n    with it if there are no tasks in the cgroup, which is no longer true\n    since a79a908fd2b0 (\"cgroup: introduce cgroup namespaces\").\n\n    When a cgroup namespace is created, it pins the css_set of the\n    creating task to use it as the root css_set of the namespace.  This\n    extra reference stays as long as the namespace is around and makes\n    \"cgroup.subtree_control\" think that the namespace root cgroup is not\n    empty even when it is and thus reject controller enables.\n\n    Fix it by making cgroup_subtree_control() walk and test emptiness of\n    each css_set instead of testing whether the list_head is empty.\n\n    While at it, update the comment of cgroup_task_count() to indicate\n    that the returned value may be higher than the number of tasks, which\n    has always been true due to temporary references and doesn\u0027t break\n    anything.\n\n    Signed-off-by: Tejun Heo \u003ctj@kernel.org\u003e\n    Reported-by: Evgeny Vereshchagin \u003cevvers@ya.ru\u003e\n    Cc: Serge E. Hallyn \u003cserge.hallyn@ubuntu.com\u003e\n    Cc: Aditya Kali \u003cadityakali@google.com\u003e\n    Cc: Eric W. Biederman \u003cebiederm@xmission.com\u003e\n    Cc: stable@vger.kernel.org # v4.6+\n    Fixes: a79a908fd2b0 (\"cgroup: introduce cgroup namespaces\")\n    Link: https://github.com/systemd/systemd/pull/3589#issuecomment-249089541\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 3fb86dba27fc70396927aac17601dbd717a60eb3\nAuthor: Andrey Vagin \u003cavagin@openvz.org\u003e\nDate:   Tue Sep 6 00:47:13 2016 -0700\n\n    kernel: add a helper to get an owning user namespace for a namespace\n\n    Return -EPERM if an owning user namespace is outside of a process\n    current user namespace.\n\n    v2: In a first version ns_get_owner returned ENOENT for init_user_ns.\n        This special cases was removed from this version. There is nothing\n        outside of init_user_ns, so we can return EPERM.\n    v3: rename ns-\u003eget_owner() to ns-\u003eowner(). get_* usually means that it\n    grabs a reference.\n\n    Acked-by: Serge Hallyn \u003cserge@hallyn.com\u003e\n    Signed-off-by: Andrei Vagin \u003cavagin@openvz.org\u003e\n    Signed-off-by: Eric W. Biederman \u003cebiederm@xmission.com\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit e544e09109574f39595e7b069df839c857c6302d\nAuthor: Seth Forshee \u003cseth.forshee@canonical.com\u003e\nDate:   Wed Sep 23 15:16:04 2015 -0500\n\n    fs: Limit file caps to the user namespace of the super block\n\n    Capability sets attached to files must be ignored except in the\n    user namespaces where the mounter is privileged, i.e. s_user_ns\n    and its descendants. Otherwise a vector exists for gaining\n    privileges in namespaces where a user is not already privileged.\n\n    Add a new helper function, current_in_user_ns(), to test whether a user\n    namespace is the same as or a descendant of another namespace.\n    Use this helper to determine whether a file\u0027s capability set\n    should be applied to the caps constructed during exec.\n\n    --EWB Replaced in_userns with the simpler current_in_userns.\n\n    Acked-by: Serge Hallyn \u003cserge.hallyn@canonical.com\u003e\n    Signed-off-by: Seth Forshee \u003cseth.forshee@canonical.com\u003e\n    Signed-off-by: Eric W. Biederman \u003cebiederm@xmission.com\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 4ec041662bc02b09e745c6b69b7891ee1478c9b1\nAuthor: Johannes Weiner \u003cjweiner@fb.com\u003e\nDate:   Mon Sep 19 14:44:38 2016 -0700\n\n    cgroup: duplicate cgroup reference when cloning sockets\n\n    When a socket is cloned, the associated sock_cgroup_data is duplicated\n    but not its reference on the cgroup.  As a result, the cgroup reference\n    count will underflow when both sockets are destroyed later on.\n\n    Fixes: bd1060a1d671 (\"sock, cgroup: add sock-\u003esk_cgroup\")\n    Link: http://lkml.kernel.org/r/20160914194846.11153-2-hannes@cmpxchg.org\n    Signed-off-by: Johannes Weiner \u003channes@cmpxchg.org\u003e\n    Acked-by: Tejun Heo \u003ctj@kernel.org\u003e\n    Cc: Michal Hocko \u003cmhocko@suse.cz\u003e\n    Cc: Vladimir Davydov \u003cvdavydov@virtuozzo.com\u003e\n    Cc: \u003cstable@vger.kernel.org\u003e\t[4.5+]\n    Signed-off-by: Andrew Morton \u003cakpm@linux-foundation.org\u003e\n    Signed-off-by: Linus Torvalds \u003ctorvalds@linux-foundation.org\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 73c0e5722e9fc6f1b7fa2058fd083ceaae2568b2\nAuthor: Tejun Heo \u003ctj@kernel.org\u003e\nDate:   Wed Aug 10 11:23:44 2016 -0400\n\n    cgroup: make cgroup_path() and friends behave in the style of strlcpy()\n\n    cgroup_path() and friends used to format the path from the end and\n    thus the resulting path usually didn\u0027t start at the start of the\n    passed in buffer.  Also, when the buffer was too small, the partial\n    result was truncated from the head rather than tail and there was no\n    way to tell how long the full path would be.  These make the functions\n    less robust and more awkward to use.\n\n    With recent updates to kernfs_path(), cgroup_path() and friends can be\n    made to behave in strlcpy() style.\n\n    * cgroup_path(), cgroup_path_ns[_locked]() and task_cgroup_path() now\n      always return the length of the full path.  If buffer is too small,\n      it contains nul terminated truncated output.\n\n    * All users updated accordingly.\n\n    v2: cgroup_path() usage in kernel/sched/debug.c converted.\n\n    Signed-off-by: Tejun Heo \u003ctj@kernel.org\u003e\n    Acked-by: Greg Kroah-Hartman \u003cgregkh@linuxfoundation.org\u003e\n    Cc: Serge Hallyn \u003cserge.hallyn@ubuntu.com\u003e\n    Cc: Peter Zijlstra \u003cpeterz@infradead.org\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n    Change-Id: I8f16f5cb47c1eae59ad7539ea3fa3af825e0a125\n\ncommit d305d57e20372eb84ec7fdad134eb7c77238bd7d\nAuthor: Eric W. Biederman \u003cebiederm@xmission.com\u003e\nDate:   Fri Jul 15 06:36:44 2016 -0500\n\n    cgroupns: Only allow creation of hierarchies in the initial cgroup namespace\n\n    Unprivileged users can\u0027t use hierarchies if they create them as they do not\n    have privilieges to the root directory.\n\n    Which means the only thing a hiearchy created by an unprivileged user\n    is good for is expanding the number of cgroup links in every css_set,\n    which is a DOS attack.\n\n    We could allow hierarchies to be created in namespaces in the initial\n    user namespace.  Unfortunately there is only a single namespace for\n    the names of heirarchies, so that is likely to create more confusion\n    than not.\n\n    So do the simple thing and restrict hiearchy creation to the initial\n    cgroup namespace.\n\n    Cc: stable@vger.kernel.org\n    Fixes: a79a908fd2b0 (\"cgroup: introduce cgroup namespaces\")\n    Signed-off-by: \"Eric W. Biederman\" \u003cebiederm@xmission.com\u003e\n    Signed-off-by: Tejun Heo \u003ctj@kernel.org\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 5fbfc280f10ea9af918a7429e831b42d81d58860\nAuthor: Eric W. Biederman \u003cebiederm@xmission.com\u003e\nDate:   Fri Jul 15 06:35:24 2016 -0500\n\n    cgroupns: Fix the locking in copy_cgroup_ns\n\n    If \"clone(CLONE_NEWCGROUP...)\" is called it results in a nice lockdep\n    valid splat.\n\n    In __cgroup_proc_write the lock ordering is:\n         cgroup_mutex -- through cgroup_kn_lock_live\n         cgroup_threadgroup_rwsem\n\n    In copy_process the guts of clone the lock ordering is:\n         cgroup_threadgroup_rwsem -- through threadgroup_change_begin\n         cgroup_mutex -- through copy_namespaces -- copy_cgroup_ns\n\n    lockdep reports some a different call chains for the first ordering of\n    cgroup_mutex and cgroup_threadgroup_rwsem but it is harder to trace.\n    This is most definitely deadlock potential under the right\n    circumstances.\n\n    Fix this by by skipping the cgroup_mutex and making the locking in\n    copy_cgroup_ns mirror the locking in cgroup_post_fork which also runs\n    during fork under the cgroup_threadgroup_rwsem.\n\n    Cc: stable@vger.kernel.org\n    Fixes: a79a908fd2b0 (\"cgroup: introduce cgroup namespaces\")\n    Signed-off-by: \"Eric W. Biederman\" \u003cebiederm@xmission.com\u003e\n    Signed-off-by: Tejun Heo \u003ctj@kernel.org\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 3fa46142f591815dbaf2c860b356661755991aab\nAuthor: Martin KaFai Lau \u003ckafai@fb.com\u003e\nDate:   Thu Jun 30 10:28:42 2016 -0700\n\n    cgroup: Add cgroup_get_from_fd\n\n    Add a helper function to get a cgroup2 from a fd.  It will be\n    stored in a bpf array (BPF_MAP_TYPE_CGROUP_ARRAY) which will\n    be introduced in the later patch.\n\n    Signed-off-by: Martin KaFai Lau \u003ckafai@fb.com\u003e\n    Cc: Alexei Starovoitov \u003cast@fb.com\u003e\n    Cc: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Cc: Tejun Heo \u003ctj@kernel.org\u003e\n    Acked-by: Tejun Heo \u003ctj@kernel.org\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 8eb20de623d2df379b3a12c3ef34f86706c7cf9c\nAuthor: Tejun Heo \u003ctj@kernel.org\u003e\nDate:   Tue Jun 21 13:06:24 2016 -0400\n\n    cgroup: allow NULL return from ss-\u003ecss_alloc()\n\n    cgroup core expected css_alloc to return an ERR_PTR value on failure\n    and caused NULL deref if it returned NULL.  It\u0027s an easy mistake to\n    make from an alloc function and there\u0027s no ambiguity in what\u0027s being\n    indicated.  Update css_create() so that it interprets NULL return from\n    css_alloc as -ENOMEM.\n\n    Signed-off-by: Tejun Heo \u003ctj@kernel.org\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 61856241adaedfc4a0ecb161a492b2c34c52d9e8\nAuthor: Johannes Weiner \u003channes@cmpxchg.org\u003e\nDate:   Fri Jun 17 12:24:27 2016 -0400\n\n    cgroup: remove unnecessary 0 check from css_from_id()\n\n    css_idr allocation starts at 1, so index 0 will never point to an\n    item. css_from_id() currently filters that before asking idr_find(),\n    but idr_find() would also just return NULL, so this is not needed.\n\n    Signed-off-by: Johannes Weiner \u003channes@cmpxchg.org\u003e\n    Signed-off-by: Tejun Heo \u003ctj@kernel.org\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit e25a3db33d61ef2d8c828928eb5176bcc726bc46\nAuthor: Johannes Weiner \u003channes@cmpxchg.org\u003e\nDate:   Fri Jun 17 12:23:59 2016 -0400\n\n    cgroup: fix idr leak for the first cgroup root\n\n    The valid cgroup hierarchy ID range includes 0, so we can\u0027t filter for\n    positive numbers when freeing it, or it\u0027ll leak the first ID. No big\n    deal, just disruptive when reading the code.\n\n    The ID is freed during error handling and when the reference count\n    hits zero, so the double-free test is not necessary; remove it.\n\n    Signed-off-by: Johannes Weiner \u003channes@cmpxchg.org\u003e\n    Signed-off-by: Tejun Heo \u003ctj@kernel.org\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 45875d6655e72fb2517b44547bc1a72847f89d1c\nAuthor: Wenwei Tao \u003cww.tao0320@gmail.com\u003e\nDate:   Fri May 13 22:59:20 2016 +0800\n\n    cgroup: remove redundant cleanup in css_create\n\n    When create css failed, before call css_free_rcu_fn, we remove the css\n    id and exit the percpu_ref, but we will do these again in\n    css_free_work_fn, so they are redundant.  Especially the css id, that\n    would cause problem if we remove it twice, since it may be assigned to\n    another css after the first remove.\n\n    tj: This was broken by two commits updating the free path without\n        synchronizing the creation failure path.  This can be easily\n        triggered by trying to create more than 64k memory cgroups.\n\n    Signed-off-by: Wenwei Tao \u003cww.tao0320@gmail.com\u003e\n    Signed-off-by: Tejun Heo \u003ctj@kernel.org\u003e\n    Cc: Vladimir Davydov \u003cvdavydov@parallels.com\u003e\n    Fixes: 9a1049da9bd2 (\"percpu-refcount: require percpu_ref to be exited explicitly\")\n    Fixes: 01e586598b22 (\"cgroup: release css-\u003eid after css_free\")\n    Cc: stable@vger.kernel.org # v3.17+\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 06f9f5f260b77cb9d9536d1bcdcfb4251a43fb6b\nAuthor: Felipe Balbi \u003cfelipe.balbi@linux.intel.com\u003e\nDate:   Thu May 12 12:34:38 2016 +0300\n\n    cgroup: fix compile warning\n\n    commit 4f41fc59620f (\"cgroup, kernfs: make mountinfo\n     show properly scoped path for cgroup namespaces\")\n     added the following compile warning:\n\n    kernel/cgroup.c: In function ‘cgroup_show_path’:\n    kernel/cgroup.c:1634:15: warning: unused variable ‘ret’ [-Wunused-variable]\n      int len \u003d 0, ret \u003d 0;\n                   ^\n    fix it.\n\n    Fixes: 4f41fc59620f (\"cgroup, kernfs: make mountinfo show properly scoped path for cgroup namespaces\")\n    Signed-off-by: Felipe Balbi \u003cfelipe.balbi@linux.intel.com\u003e\n    Signed-off-by: Tejun Heo \u003ctj@kernel.org\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 3887afe1843a8e67b65ba79803ece0488b76c3d9\nAuthor: Serge E. Hallyn \u003cserge.hallyn@ubuntu.com\u003e\nDate:   Mon May 9 09:59:55 2016 -0500\n\n    cgroup, kernfs: make mountinfo show properly scoped path for cgroup namespaces\n\n    Patch summary:\n\n    When showing a cgroupfs entry in mountinfo, show the path of the mount\n    root dentry relative to the reader\u0027s cgroup namespace root.\n\n    Short explanation (courtesy of mkerrisk):\n\n    If we create a new cgroup namespace, then we want both /proc/self/cgroup\n    and /proc/self/mountinfo to show cgroup paths that are correctly\n    virtualized with respect to the cgroup mount point.  Previous to this\n    patch, /proc/self/cgroup shows the right info, but /proc/self/mountinfo\n    does not.\n\n    Long version:\n\n    When a uid 0 task which is in freezer cgroup /a/b, unshares a new cgroup\n    namespace, and then mounts a new instance of the freezer cgroup, the new\n    mount will be rooted at /a/b.  The root dentry field of the mountinfo\n    entry will show \u0027/a/b\u0027.\n\n     cat \u003e /tmp/do1 \u003c\u003c EOF\n     mount -t cgroup -o freezer freezer /mnt\n     grep freezer /proc/self/mountinfo\n     EOF\n\n     unshare -Gm  bash /tmp/do1\n     \u003e 330 160 0:34 / /sys/fs/cgroup/freezer rw,nosuid,nodev,noexec,relatime - cgroup cgroup rw,freezer\n     \u003e 355 133 0:34 /a/b /mnt rw,relatime - cgroup freezer rw,freezer\n\n    The task\u0027s freezer cgroup entry in /proc/self/cgroup will simply show\n    \u0027/\u0027:\n\n     grep freezer /proc/self/cgroup\n     9:freezer:/\n\n    If instead the same task simply bind mounts the /a/b cgroup directory,\n    the resulting mountinfo entry will again show /a/b for the dentry root.\n    However in this case the task will find its own cgroup at /mnt/a/b,\n    not at /mnt:\n\n     mount --bind /sys/fs/cgroup/freezer/a/b /mnt\n     130 25 0:34 /a/b /mnt rw,nosuid,nodev,noexec,relatime shared:21 - cgroup cgroup rw,freezer\n\n    In other words, there is no way for the task to know, based on what is\n    in mountinfo, which cgroup directory is its own.\n\n    Example (by mkerrisk):\n\n    First, a little script to save some typing and verbiage:\n\n    echo -e \"\\t/proc/self/cgroup:\\t$(cat /proc/self/cgroup | grep freezer)\"\n    cat /proc/self/mountinfo | grep freezer |\n            awk \u0027{print \"\\tmountinfo:\\t\\t\" $4 \"\\t\" $5}\u0027\n\n    Create cgroup, place this shell into the cgroup, and look at the state\n    of the /proc files:\n\n    2653\n    2653                         # Our shell\n    14254                        # cat(1)\n            /proc/self/cgroup:      10:freezer:/a/b\n            mountinfo:              /       /sys/fs/cgroup/freezer\n\n    Create a shell in new cgroup and mount namespaces. The act of creating\n    a new cgroup namespace causes the process\u0027s current cgroups directories\n    to become its cgroup root directories. (Here, I\u0027m using my own version\n    of the \"unshare\" utility, which takes the same options as the util-linux\n    version):\n\n    Look at the state of the /proc files:\n\n            /proc/self/cgroup:      10:freezer:/\n            mountinfo:              /       /sys/fs/cgroup/freezer\n\n    The third entry in /proc/self/cgroup (the pathname of the cgroup inside\n    the hierarchy) is correctly virtualized w.r.t. the cgroup namespace, which\n    is rooted at /a/b in the outer namespace.\n\n    However, the info in /proc/self/mountinfo is not for this cgroup\n    namespace, since we are seeing a duplicate of the mount from the\n    old mount namespace, and the info there does not correspond to the\n    new cgroup namespace. However, trying to create a new mount still\n    doesn\u0027t show us the right information in mountinfo:\n\n                                          # propagating to other mountns\n            /proc/self/cgroup:      7:freezer:/\n            mountinfo:              /a/b    /mnt/freezer\n\n    The act of creating a new cgroup namespace caused the process\u0027s\n    current freezer directory, \"/a/b\", to become its cgroup freezer root\n    directory. In other words, the pathname directory of the directory\n    within the newly mounted cgroup filesystem should be \"/\",\n    but mountinfo wrongly shows us \"/a/b\". The consequence of this is\n    that the process in the cgroup namespace cannot correctly construct\n    the pathname of its cgroup root directory from the information in\n    /proc/PID/mountinfo.\n\n    With this patch, the dentry root field in mountinfo is shown relative\n    to the reader\u0027s cgroup namespace.  So the same steps as above:\n\n            /proc/self/cgroup:      10:freezer:/a/b\n            mountinfo:              /       /sys/fs/cgroup/freezer\n            /proc/self/cgroup:      10:freezer:/\n            mountinfo:              /../..  /sys/fs/cgroup/freezer\n            /proc/self/cgroup:      10:freezer:/\n            mountinfo:              /       /mnt/freezer\n\n    cgroup.clone_children  freezer.parent_freezing  freezer.state      tasks\n    cgroup.procs           freezer.self_freezing    notify_on_release\n    3164\n    2653                   # First shell that placed in this cgroup\n    3164                   # Shell started by \u0027unshare\u0027\n    14197                  # cat(1)\n\n    Signed-off-by: Serge Hallyn \u003cserge.hallyn@ubuntu.com\u003e\n    Tested-by: Michael Kerrisk \u003cmtk.manpages@gmail.com\u003e\n    Acked-by: Michael Kerrisk \u003cmtk.manpages@gmail.com\u003e\n    Signed-off-by: Tejun Heo \u003ctj@kernel.org\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 7b19bf36d0103807b0f478e5523c6f6e539a781a\nAuthor: Tejun Heo \u003ctj@kernel.org\u003e\nDate:   Tue Mar 8 11:51:26 2016 -0500\n\n    cgroup: implement cgroup_subsys-\u003eimplicit_on_dfl\n\n    Some controllers, perf_event for now and possibly freezer in the\n    future, don\u0027t really make sense to control explicitly through\n    \"cgroup.subtree_control\".  For example, the primary role of perf_event\n    is identifying the cgroups of tasks; however, because the controller\n    also keeps a small amount of state per cgroup, it can\u0027t be replaced\n    with simple cgroup membership tests.\n\n    This patch implements cgroup_subsys-\u003eimplicit_on_dfl flag.  When set,\n    the controller is implicitly enabled on all cgroups on the v2\n    hierarchy so that utility type controllers such as perf_event can be\n    enabled and function transparently.\n\n    An implicit controller doesn\u0027t show up in \"cgroup.controllers\" or\n    \"cgroup.subtree_control\", is exempt from no internal process rule and\n    can be stolen from the default hierarchy even if there are non-root\n    csses.\n\n    v2: Reimplemented on top of the recent updates to css handling and\n        subsystem rebinding.  Rebinding implicit subsystems is now a\n        simple matter of exempting it from the busy subsystem check.\n\n    Signed-off-by: Tejun Heo \u003ctj@kernel.org\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit f67ca4f794d2a49c7ee498716012ac958dd843d0\nAuthor: Tejun Heo \u003ctj@kernel.org\u003e\nDate:   Tue Mar 8 11:51:26 2016 -0500\n\n    cgroup: use css_set-\u003emg_dst_cgrp for the migration target cgroup\n\n    Migration can be multi-target on the default hierarchy when a\n    controller is enabled - processes belonging to each child cgroup have\n    to be moved to the child cgroup itself to refresh css association.\n\n    This isn\u0027t a problem for cgroup_migrate_add_src() as each source\n    css_set still maps to single source and target cgroups; however,\n    cgroup_migrate_prepare_dst() is called once after all source css_sets\n    are added and thus might not have a single destination cgroup.  This\n    is currently worked around by specifying NULL for @dst_cgrp and using\n    the source\u0027s default cgroup as destination as the only multi-target\n    migration in use is self-targetting.  While this works, it\u0027s subtle\n    and clunky.\n\n    As all taget cgroups are already specified while preparing the source\n    css_sets, this clunkiness can easily be removed by recording the\n    target cgroup in each source css_set.  This patch adds\n    css_set-\u003emg_dst_cgrp which is recorded on cgroup_migrate_src() and\n    used by cgroup_migrate_prepare_dst().  This also makes migration code\n    ready for arbitrary multi-target migration.\n\n    Signed-off-by: Tejun Heo \u003ctj@kernel.org\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 5a298565a2b491136dba1f06e753eb37d1cf5c77\nAuthor: Tejun Heo \u003ctj@kernel.org\u003e\nDate:   Tue Mar 8 11:51:26 2016 -0500\n\n    cgroup: make cgroup[_taskset]_migrate() take cgroup_root instead of cgroup\n\n    On the default hierarchy, a migration can be multi-source and/or\n    multi-destination.  cgroup_taskest_migrate() used to incorrectly\n    assume single destination cgroup but the bug has been fixed by\n    1f7dd3e5a6e4 (\"cgroup: fix handling of multi-destination migration\n    from subtree_control enabling\").\n\n    Since the commit, @dst_cgrp to cgroup[_taskset]_migrate() is only used\n    to determine which subsystems are affected or which cgroup_root the\n    migration is taking place in.  As such, @dst_cgrp is misleading.  This\n    patch replaces @dst_cgrp with @root.\n\n    Signed-off-by: Tejun Heo \u003ctj@kernel.org\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 5c9ddbc98a00396e6f984563db930519f8af27a7\nAuthor: Tejun Heo \u003ctj@kernel.org\u003e\nDate:   Tue Mar 8 11:51:25 2016 -0500\n\n    cgroup: move migration destination verification out of cgroup_migrate_prepare_dst()\n\n    cgroup_migrate_prepare_dst() verifies whether the destination cgroup\n    is allowable; however, the test doesn\u0027t really belong there.  It\u0027s too\n    deep and common in the stack and as a result the test itself is gated\n    by another test.\n\n    Separate the test out into cgroup_may_migrate_to() and update\n    cgroup_attach_task() and cgroup_transfer_tasks() to perform the test\n    directly.  This doesn\u0027t cause any behavior differences.\n\n    Signed-off-by: Tejun Heo \u003ctj@kernel.org\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 5a3d735de7547f062bf71c10299fe3461b3f3bf1\nAuthor: Tejun Heo \u003ctj@kernel.org\u003e\nDate:   Tue Mar 8 11:51:25 2016 -0500\n\n    cgroup: fix incorrect destination cgroup in cgroup_update_dfl_csses()\n\n    cgroup_update_dfl_csses() should move each task in the subtree to\n    self; however, it was incorrectly calling cgroup_migrate_add_src()\n    with the root of the subtree as @dst_cgrp.  Fortunately,\n    cgroup_migrate_add_src() currently uses @dst_cgrp only to determine\n    the hierarchy and the bug doesn\u0027t cause any actual breakages.  Fix it.\n\n    Signed-off-by: Tejun Heo \u003ctj@kernel.org\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit f9017131600b76950759247ff5d93f7b50873d2c\nAuthor: Tejun Heo \u003ctj@kernel.org\u003e\nDate:   Thu Mar 3 09:58:01 2016 -0500\n\n    cgroup: update css iteration in cgroup_update_dfl_csses()\n\n    The existing sequences of operations ensure that the offlining csses\n    are drained before cgroup_update_dfl_csses(), so even though\n    cgroup_update_dfl_csses() uses css_for_each_descendant_pre() to walk\n    the target cgroups, it doesn\u0027t end up operating on dead cgroups.\n    Also, the function explicitly excludes the subtree root from\n    operation.\n\n    This is fragile and inconsistent with the rest of css update\n    operations.  This patch updates cgroup_update_dfl_csses() to use\n    cgroup_for_each_live_descendant_pre() instead and include the subtree\n    root.\n\n    Signed-off-by: Tejun Heo \u003ctj@kernel.org\u003e\n    Acked-by: Zefan Li \u003clizefan@huawei.com\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 3f82ae7e0edcddf742a23975cbb6ff9d97c11014\nAuthor: Tejun Heo \u003ctj@kernel.org\u003e\nDate:   Thu Mar 3 09:58:01 2016 -0500\n\n    cgroup: allocate 2x cgrp_cset_links when setting up a new root\n\n    During prep, cgroup_setup_root() allocates cgrp_cset_links matching\n    the number of existing css_sets to later link the new root.  This is\n    fine for now as the only operation which can happen inbetween is\n    rebind_subsystems() and rebinding of empty subsystems doesn\u0027t create\n    new css_sets.\n\n    However, while not yet allowed, with the recent reimplementation,\n    rebind_subsystems() can rebind subsystems with descendant csses and\n    thus can create new css_sets.  This patch makes cgroup_setup_root()\n    allocate 2x of the existing css_sets so that later use of live\n    subsystem rebinding doesn\u0027t blow up.\n\n    Signed-off-by: Tejun Heo \u003ctj@kernel.org\u003e\n    Acked-by: Zefan Li \u003clizefan@huawei.com\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 6faa69554235d709a0ccc86963f5abce29848701\nAuthor: Tejun Heo \u003ctj@kernel.org\u003e\nDate:   Thu Mar 3 09:58:01 2016 -0500\n\n    cgroup: make cgroup_calc_subtree_ss_mask() take @this_ss_mask\n\n    cgroup_calc_subtree_ss_mask() currently takes @cgrp and\n    @subtree_control.  @cgrp is used for two purposes - to decide whether\n    it\u0027s for default hierarchy and the mask of available subsystems.  The\n    former doesn\u0027t matter as the results are the same regardless.  The\n    latter can be specified directly through a subsystem mask.\n\n    This patch makes cgroup_calc_subtree_ss_mask() perform the same\n    calculations for both default and legacy hierarchies and take\n    @this_ss_mask for available subsystems.  @cgrp is no longer used and\n    dropped.  This is to allow using the function in contexts where\n    available controllers can\u0027t be decided from the cgroup.\n\n    v2: cgroup_refres_subtree_ss_mask() is removed by a previous patch.\n        Updated accordingly.\n\n    Signed-off-by: Tejun Heo \u003ctj@kernel.org\u003e\n    Acked-by: Zefan Li \u003clizefan@huawei.com\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit a6fa3a1ebe3b4873bb8afaf9078ab998eaf357ea\nAuthor: Tejun Heo \u003ctj@kernel.org\u003e\nDate:   Thu Mar 3 09:58:01 2016 -0500\n\n    cgroup: reimplement rebind_subsystems() using cgroup_apply_control() and friends\n\n    rebind_subsystem() open codes quite a bit of css and interface file\n    manipulations.  It tries to be fail-safe but doesn\u0027t quite achieve it.\n    It can be greatly simplified by using the new css management helpers.\n    This patch reimplements rebind_subsytsems() using\n    cgroup_apply_control() and friends.\n\n    * The half-baked rollback on file creation failure is dropped.  It is\n      an extremely cold path, failure isn\u0027t critical, and, aside from\n      kernel bugs, the only reason it can fail is memory allocation\n      failure which pretty much doesn\u0027t happen for small allocations.\n\n    * As cgroup_apply_control_disable() is now used to clean up root\n      cgroup on rebind, make sure that it doesn\u0027t end up killing root\n      csses.\n\n    * All callers of rebind_subsystems() are updated to use\n      cgroup_lock_and_drain_offline() as the apply_control functions\n      require drained subtree.\n\n    * This leaves cgroup_refresh_subtree_ss_mask() without any user.\n      Removed.\n\n    * css_populate_dir() and css_clear_dir() no longer needs\n      @cgrp_override parameter.  Dropped.\n\n    * While at it, add WARN_ON() to rebind_subsystem() calls which are\n      expected to always succeed just in case.\n\n    While the rules visible to userland aren\u0027t changed, this\n    reimplementation not only simplifies rebind_subsystems() but also\n    allows it to disable and enable csses recursively.  This can be used\n    to implement more flexible rebinding.\n\n    Signed-off-by: Tejun Heo \u003ctj@kernel.org\u003e\n    Acked-by: Zefan Li \u003clizefan@huawei.com\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit f640e4152984f8035a5f4d6b6cfe9ff42f7689ce\nAuthor: Tejun Heo \u003ctj@kernel.org\u003e\nDate:   Thu Mar 3 09:58:00 2016 -0500\n\n    cgroup: use cgroup_apply_enable_control() in cgroup creation path\n\n    cgroup_create() manually updates control masks and creates child csses\n    which cgroup_mkdir() then manually populates.  Both can be simplified\n    by using cgroup_apply_enable_control() and friends.  The only catch is\n    that it calls css_populate_dir() with NULL cgroup-\u003ekn during\n    cgroup_create().  This is worked around by making the function noop on\n    NULL kn.\n\n    Signed-off-by: Tejun Heo \u003ctj@kernel.org\u003e\n    Acked-by: Zefan Li \u003clizefan@huawei.com\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit c54842ae72c0a2ac5817af6a8b35b414aa277fe1\nAuthor: Tejun Heo \u003ctj@kernel.org\u003e\nDate:   Thu Mar 3 09:58:00 2016 -0500\n\n    cgroup: combine cgroup_mutex locking and offline css draining\n\n    cgroup_drain_offline() is used to wait for csses being offlined to\n    uninstall itself from cgroup-\u003esubsys[] array so that new csses can be\n    installed.  The function\u0027s only user, cgroup_subtree_control_write(),\n    calls it after performing some checks and restarts the whole process\n    via restart_syscall() if draining has to release cgroup_mutex to wait.\n\n    This can be simplified by draining before other synchronized\n    operations so that there\u0027s nothing to restart.  This patch converts\n    cgroup_drain_offline() to cgroup_lock_and_drain_offline() which\n    performs both locking and draining and updates cgroup_kn_lock_live()\n    use it instead of cgroup_mutex() if requested.  This combined locking\n    and draining operations are easier to use and less error-prone.\n\n    While at it, add WARNs in control_apply functions which triggers if\n    the subtree isn\u0027t properly drained.\n\n    Signed-off-by: Tejun Heo \u003ctj@kernel.org\u003e\n    Acked-by: Zefan Li \u003clizefan@huawei.com\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 625cd6484bc5cff232fab1cb45d29262ee9f5729\nAuthor: Tejun Heo \u003ctj@kernel.org\u003e\nDate:   Thu Mar 3 09:58:00 2016 -0500\n\n    cgroup: factor out cgroup_{apply|finalize}_control() from cgroup_subtree_control_write()\n\n    Factor out cgroup_{apply|finalize}_control() so that control mask\n    update can be done in several simple steps.  This patch doesn\u0027t\n    introduce behavior changes.\n\n    Signed-off-by: Tejun Heo \u003ctj@kernel.org\u003e\n    Acked-by: Zefan Li \u003clizefan@huawei.com\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 5bd05286da6f63d68ddcfb56cdf995ff1adf4ed3\nAuthor: Tejun Heo \u003ctj@kernel.org\u003e\nDate:   Thu Mar 3 09:57:59 2016 -0500\n\n    cgroup: introduce cgroup_{save|propagate|restore}_control()\n\n    While controllers are being enabled and disabled in\n    cgroup_subtree_control_write(), the original subsystem masks are\n    stashed in local variables so that they can be restored if the\n    operation fails in the middle.\n\n    This patch adds dedicated fields to struct cgroup to be used instead\n    of the local variables and implements functions to stash the current\n    values, propagate the changes and restore them recursively.  Combined\n    with the previous changes, this makes subsystem management operations\n    fully recursive and modularlized.  This will be used to expand cgroup\n    core functionalities.\n\n    While at it, remove now unused @css_enable and @css_disable from\n    cgroup_subtree_control_write().\n\n    Signed-off-by: Tejun Heo \u003ctj@kernel.org\u003e\n    Acked-by: Zefan Li \u003clizefan@huawei.com\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 4f1ca466ce38a618d6faa0119ea3cbbf2e45b582\nAuthor: Tejun Heo \u003ctj@kernel.org\u003e\nDate:   Thu Mar 3 09:57:59 2016 -0500\n\n    cgroup: make cgroup_drain_offline() and cgroup_apply_control_{disable|enable}() recursive\n\n    The three factored out css management operations -\n    cgroup_drain_offline() and cgroup_apply_control_{disable|enable}() -\n    only depend on the current state of the target cgroups and idempotent\n    and thus can be easily made to operate on the subtree instead of the\n    immediate children.\n\n    This patch introduces the iterators which walk live subtree and\n    converts the three functions to operate on the subtree including self\n    instead of the children.  While this leads to spurious walking and be\n    slightly more expensive, it will allow them to be used for wider scope\n    of operations.\n\n    Note that cgroup_drain_offline() now tests for whether a css is dying\n    before trying to drain it.  This is to avoid trying to drain live\n    csses as there can be mix of live and dying csses in a subtree unlike\n    children of the same parent.\n\n    Signed-off-by: Tejun Heo \u003ctj@kernel.org\u003e\n    Acked-by: Zefan Li \u003clizefan@huawei.com\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 1618a596e9d8624aa4dca92a9b1b8d3ea02523f4\nAuthor: Tejun Heo \u003ctj@kernel.org\u003e\nDate:   Thu Mar 3 09:57:59 2016 -0500\n\n    cgroup: factor out cgroup_apply_control_enable() from cgroup_subtree_control_write()\n\n    Factor out css enabling and showing into cgroup_apply_control_enable().\n\n    * Nest subsystem walk inside child walk.  The child walk will later be\n      converted to subtree walk which is a bit more expensive.\n\n    * Instead of operating on the differential masks @css_enable, simply\n      enable or show csses according to the current cgroup_control() and\n      cgroup_ss_mask().  This leads to the same result and is simpler and\n      more robust.\n\n    Signed-off-by: Tejun Heo \u003ctj@kernel.org\u003e\n    Acked-by: Zefan Li \u003clizefan@huawei.com\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit a1a9db1ad261c6cd33aad5980aaad3a985957bd5\nAuthor: Tejun Heo \u003ctj@kernel.org\u003e\nDate:   Thu Mar 3 09:57:59 2016 -0500\n\n    cgroup: factor out cgroup_apply_control_disable() from cgroup_subtree_control_write()\n\n    Factor out css disabling and hiding into cgroup_apply_control_disable().\n\n    * Nest subsystem walk inside child walk.  The child walk will later be\n      converted to subtree walk which is a bit more expensive.\n\n    * Instead of operating on the differential masks @css_enable and\n      @css_disable, simply disable or hide csses according to the current\n      cgroup_control() and cgroup_ss_mask().  This leads to the same\n      result and is simpler and more robust.\n\n    * This allows error handling path to share the same code.\n\n    Signed-off-by: Tejun Heo \u003ctj@kernel.org\u003e\n    Acked-by: Zefan Li \u003clizefan@huawei.com\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 59c06cb28d873b12b6e161366783b318db208912\nAuthor: Tejun Heo \u003ctj@kernel.org\u003e\nDate:   Thu Mar 3 09:57:59 2016 -0500\n\n    cgroup: factor out cgroup_drain_offline() from cgroup_subtree_control_write()\n\n    Factor out async css offline draining into cgroup_drain_offline().\n\n    * Nest subsystem walk inside child walk.  The child walk will later be\n      converted to subtree walk which is a bit more expensive.\n\n    * Relocate the draining above subsystem mask preparation, which\n      doesn\u0027t create any behavior differences but helps further\n      refactoring.\n\n    Signed-off-by: Tejun Heo \u003ctj@kernel.org\u003e\n    Acked-by: Zefan Li \u003clizefan@huawei.com\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 4083639cd15ab9941a5cb65ef04ce325203e124f\nAuthor: Tejun Heo \u003ctj@kernel.org\u003e\nDate:   Thu Mar 3 09:57:58 2016 -0500\n\n    cgroup: introduce cgroup_control() and cgroup_ss_mask()\n\n    When a controller is enabled and visible on a non-root cgroup is\n    determined by subtree_control and subtree_ss_mask of the parent\n    cgroup.  For a root cgroup, by the type of the hierarchy and which\n    controllers are attached to it.  Deciding the above on each usage is\n    fragile and unnecessarily complicates the users.\n\n    This patch introduces cgroup_control() and cgroup_ss_mask() which\n    calculate and return the [visibly] enabled subsyste mask for the\n    specified cgroup and conver the existing usages.\n\n    * cgroup_e_css() is restructured for simplicity.\n\n    * cgroup_calc_subtree_ss_mask() and cgroup_subtree_control_write() no\n      longer need to distinguish root and non-root cases.\n\n    * With cgroup_control(), cgroup_controllers_show() can now handle both\n      root and non-root cases.  cgroup_root_controllers_show() is removed.\n\n    v2: cgroup_control() updated to yield the correct result on v1\n        hierarchies too.  cgroup_subtree_control_write() converted.\n\n    Signed-off-by: Tejun Heo \u003ctj@kernel.org\u003e\n    Acked-by: Zefan Li \u003clizefan@huawei.com\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 7fd605c7bc679e356623744469b00f094ecf7729\nAuthor: Tejun Heo \u003ctj@kernel.org\u003e\nDate:   Thu Mar 3 09:57:58 2016 -0500\n\n    cgroup: factor out cgroup_create() out of cgroup_mkdir()\n\n    We\u0027re in the process of refactoring cgroup and css management paths to\n    separate them out to eventually allow cgroups which aren\u0027t visible\n    through cgroup fs.  This patch factors out cgroup_create() out of\n    cgroup_mkdir().  cgroup_create() contains all internal object creation\n    and initialization.  cgroup_mkdir() uses cgroup_create() to create the\n    internal cgroup and adds interface directory and file creation.\n\n    This patch doesn\u0027t cause any behavior differences.\n\n    Signed-off-by: Tejun Heo \u003ctj@kernel.org\u003e\n    Acked-by: Zefan Li \u003clizefan@huawei.com\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 7123b2ea7428e057a3192c18a20ba6455699491c\nAuthor: Tejun Heo \u003ctj@kernel.org\u003e\nDate:   Thu Mar 3 09:57:58 2016 -0500\n\n    cgroup: reorder operations in cgroup_mkdir()\n\n    Currently, operations to initialize internal objects and create\n    interface directory and files are intermixed in cgroup_mkdir().  We\u0027re\n    in the process of refactoring cgroup and css management paths to\n    separate them out to eventually allow cgroups which aren\u0027t visible\n    through cgroup fs.\n\n    This patch reorders operations inside cgroup_mkdir() so that interface\n    directory and file handling comes after internal object\n    initialization.  This will enable further refactoring.\n\n    Signed-off-by: Tejun Heo \u003ctj@kernel.org\u003e\n    Acked-by: Zefan Li \u003clizefan@huawei.com\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 2066c2c2ede38412e445a4d66388513f583deca4\nAuthor: Tejun Heo \u003ctj@kernel.org\u003e\nDate:   Thu Mar 3 09:57:58 2016 -0500\n\n    cgroup: explicitly track whether a cgroup_subsys_state is visible to userland\n\n    Currently, whether a css (cgroup_subsys_state) has its interface files\n    created is not tracked and assumed to change together with the owning\n    cgroup\u0027s lifecycle.  cgroup directory and interface creation is being\n    separated out from internal object creation to help refactoring and\n    eventually allow cgroups which are not visible through cgroupfs.\n\n    This patch adds CSS_VISIBLE to track whether a css has its interface\n    files created and perform management operations only when necessary\n    which helps decoupling interface file handling from internal object\n    lifecycle.  After this patch, all css interface file management\n    functions can be called regardless of the current state and will\n    achieve the expected result.\n\n    Signed-off-by: Tejun Heo \u003ctj@kernel.org\u003e\n    Acked-by: Zefan Li \u003clizefan@huawei.com\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit b0318e7d530f2047434e44b4b48f1c79bb3f1bb6\nAuthor: Tejun Heo \u003ctj@kernel.org\u003e\nDate:   Thu Mar 3 09:57:58 2016 -0500\n\n    cgroup: separate out interface file creation from css creation\n\n    Currently, interface files are created when a css is created depending\n    on whether @visible is set.  This patch separates out the two into\n    separate steps to help code refactoring and eventually allow cgroups\n    which aren\u0027t visible through cgroup fs.\n\n    Move css_populate_dir() out of create_css() and drop @visible.  While\n    at it, rename the function to css_create() for consistency.\n\n    Signed-off-by: Tejun Heo \u003ctj@kernel.org\u003e\n    Acked-by: Zefan Li \u003clizefan@huawei.com\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit f801c6eebaa5354f4d2e0af0c841410f409a7641\nAuthor: Tejun Heo \u003ctj@kernel.org\u003e\nDate:   Thu Mar 3 09:57:57 2016 -0500\n\n    cgroup: suppress spurious de-populated events\n\n    During task migration, tasks may transfer between two css_sets which\n    are associated with the same cgroup.  If those tasks are the only\n    tasks in the cgroup, this currently triggers a spurious de-populated\n    event on the cgroup.\n\n    Fix it by bumping up populated count before bumping it down during\n    migration to ensure that it doesn\u0027t reach zero spuriously.\n\n    Signed-off-by: Tejun Heo \u003ctj@kernel.org\u003e\n    Acked-by: Zefan Li \u003clizefan@huawei.com\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit dd24f6c8c3ee76bb1941880bf569b7051d8b03c1\nAuthor: Tejun Heo \u003ctj@kernel.org\u003e\nDate:   Thu Mar 3 09:57:57 2016 -0500\n\n    cgroup: re-hash init_css_set after subsystems are initialized\n\n    css_sets are hashed by their subsys[] contents and in cgroup_init()\n    init_css_set is hashed early, before subsystem inits, when all entries\n    in its subsys[] are NULL, so that cgroup_dfl_root initialization can\n    find and link to it.  As subsystems are initialized,\n    init_css_set.subsys[] is filled up but the hashing is never updated\n    making init_css_set hashed in the wrong place.  While incorrect, this\n    doesn\u0027t cause a critical failure as css_set management code would\n    create an identical css_set dynamically.\n\n    Fix it by rehashing init_css_set after subsystems are initialized.\n    While at it, drop unnecessary @key local variable.\n\n    Signed-off-by: Tejun Heo \u003ctj@kernel.org\u003e\n    Acked-by: Zefan Li \u003clizefan@huawei.com\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 2744bd880e5ff4aa7541856f1c6ab24e684bfd26\nAuthor: Vladimir Davydov \u003cvdavydov@virtuozzo.com\u003e\nDate:   Tue Mar 1 19:56:30 2016 +0300\n\n    cgroup: reset css on destruction\n\n    An associated css can be around for quite a while after a cgroup\n    directory has been removed. In general, it makes sense to reset it to\n    defaults so as not to worry about any remnants. For instance, memory\n    cgroup needs to reset memory.low, otherwise pages charged to a dead\n    cgroup might never get reclaimed. There\u0027s -\u003ecss_reset callback, which\n    would fit perfectly for the purpose. Currently, it\u0027s only called when a\n    subsystem is disabled in the unified hierarchy and there are other\n    subsystems dependant on it. Let\u0027s call it on css destruction as well.\n\n    Suggested-by: Johannes Weiner \u003channes@cmpxchg.org\u003e\n    Signed-off-by: Vladimir Davydov \u003cvdavydov@virtuozzo.com\u003e\n    Signed-off-by: Tejun Heo \u003ctj@kernel.org\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit f5bd5544b6bddb53f7b5d308484217fbc45aa33e\nAuthor: Tejun Heo \u003ctj@kernel.org\u003e\nDate:   Sun Feb 28 08:59:33 2016 -0500\n\n    cgroup: fix and restructure error handling in copy_cgroup_ns()\n\n    copy_cgroup_ns()\u0027s error handling was broken and the attempt to fix it\n    d22025570e2e (\"cgroup: fix alloc_cgroup_ns() error handling in\n    copy_cgroup_ns()\") was broken too in that it ended up trying an\n    ERR_PTR() value.\n\n    There\u0027s only one place where copy_cgroup_ns() needs to perform cleanup\n    after failure.  Simplify and fix the error handling by removing the\n    goto\u0027s.\n\n    Signed-off-by: Tejun Heo \u003ctj@kernel.org\u003e\n    Reported-by: Dan Carpenter \u003cdan.carpenter@oracle.com\u003e\n    Acked-by: Serge E. Hallyn \u003cserge.hallyn@ubuntu.com\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 4c9059210bbb9f70f40477b4646a068d64f6ba5a\nAuthor: Xiubo Li \u003clixiubo@cmss.chinamobile.com\u003e\nDate:   Fri Feb 26 13:07:38 2016 +0800\n\n    cgroup: fix a mistake in warning message\n\n    There is a mistake about the print format name:id \u003c--\u003e %d:%s, which\n    the name is \u0027char *\u0027 type and id is \u0027int\u0027 type.  Change \"name:id\" to\n    \"id:name\" instead to be consistent with \"cgroup_subsys %d:%s\".\n\n    Signed-off-by: Xiubo Li \u003clixiubo@cmss.chinamobile.com\u003e\n    Acked-by: Zefan Li \u003clizefan@huawei.com\u003e\n    Signed-off-by: Tejun Heo \u003ctj@kernel.org\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 4186b3fe864626b700ff6f88bb87d17be7d2f62e\nAuthor: Tejun Heo \u003ctj@kernel.org\u003e\nDate:   Tue Feb 23 10:00:51 2016 -0500\n\n    cgroup: use -\u003esubtree_control when testing no internal process rule\n\n    No internal process rule is enforced by cgroup_migrate_prepare_dst()\n    during process migration.  It tests whether the target cgroup\u0027s\n    -\u003echild_subsys_mask is zero which is different from \"subtree_control\"\n    write path which tests -\u003esubtree_control.  This hasn\u0027t mattered\n    because up until now, both -\u003echild_subsys_mask and -\u003esubtree_control\n    are zero or non-zero at the same time.  However, with the planned\n    addition of implicit controllers, this will no longer be true.\n\n    This patch prepares for the change by making\n    cgorup_migrate_prepare_dst() test -\u003esubtree_control instead.\n\n    Signed-off-by: Tejun Heo \u003ctj@kernel.org\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit b2318f76a5ebd42e4655cbdbfebab7c3aeef9355\nAuthor: Tejun Heo \u003ctj@kernel.org\u003e\nDate:   Tue Feb 23 10:00:51 2016 -0500\n\n    cgroup: make css_tryget_online_from_dir() also recognize cgroup2 fs\n\n    The function currently returns -EBADF for a directory on the default\n    hierarchy.  Make it also recognize cgroup2_fs_type.  This will be used\n    for perf_event cgroup2 support.\n\n    Signed-off-by: Tejun Heo \u003ctj@kernel.org\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 32ed0cca13bfabfd89a912fd0713d244d919c0fa\nAuthor: Tejun Heo \u003ctj@kernel.org\u003e\nDate:   Tue Feb 23 10:00:50 2016 -0500\n\n    cgroup: s/cgrp_dfl_root_/cgrp_dfl_/\n\n    These var names are unnecessarily unwiedly and another similar\n    variable will be added.  Let\u0027s shorten them.\n\n    Signed-off-by: Tejun Heo \u003ctj@kernel.org\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 0a98d3462642e0a257a6566a1398190c96ec97aa\nAuthor: Tejun Heo \u003ctj@kernel.org\u003e\nDate:   Mon Feb 22 22:25:47 2016 -0500\n\n    cgroup: make cgroup subsystem masks u16\n\n    After the recent do_each_subsys_mask() conversion, there\u0027s no reason\n    to use ulong for subsystem masks.  We\u0027ll be adding more subsystem\n    masks to persistent data structures, let\u0027s reduce its size to u16\n    which should be enough for now and the foreseeable future.\n\n    This doesn\u0027t create any noticeable behavior differences.\n\n    v2: Johannes spotted that the initial patch missed cgroup_no_v1_mask.\n        Converted.\n\n    Signed-off-by: Tejun Heo \u003ctj@kernel.org\u003e\n    Acked-by: Johannes Weiner \u003channes@cmpxchg.org\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 0b90c5a2978b31d2b28114581c5667a7cdd4948a\nAuthor: Tejun Heo \u003ctj@kernel.org\u003e\nDate:   Mon Feb 22 22:25:46 2016 -0500\n\n    cgroup: use do_each_subsys_mask() where applicable\n\n    There are several places in cgroup_subtree_control_write() which can\n    use do_each_subsys_mask() instead of manual mask testing.  Use it.\n\n    No functional changes.\n\n    Signed-off-by: Tejun Heo \u003ctj@kernel.org\u003e\n    Acked-by: Johannes Weiner \u003channes@cmpxchg.org\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 346d2aba56ce19e4a602f572263f33198353eb20\nAuthor: Tejun Heo \u003ctj@kernel.org\u003e\nDate:   Mon Feb 22 22:25:46 2016 -0500\n\n    cgroup: convert for_each_subsys_which() to do-while style\n\n    for_each_subsys_which() allows iterating subsystems specified in a\n    subsystem bitmask; unfortunately, it requires the mask to be an\n    unsigned long l-value which can be inconvenient and makes it awkward\n    to use a smaller type for subsystem masks.\n\n    This patch converts for_each_subsy_which() to do-while style which\n    allows it to drop the l-value requirement.  The new iterator is named\n    do_each_subsys_mask() / while_each_subsys_mask().\n\n    Signed-off-by: Tejun Heo \u003ctj@kernel.org\u003e\n    Cc: Aleksa Sarai \u003ccyphar@cyphar.com\u003e\n    Acked-by: Johannes Weiner \u003channes@cmpxchg.org\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit a4b5f7dd33516068b46815292640efd74d477ee4\nAuthor: Tejun Heo \u003ctj@kernel.org\u003e\nDate:   Mon Feb 22 22:25:46 2016 -0500\n\n    cgroup: s/child_subsys_mask/subtree_ss_mask/\n\n    For consistency with cgroup-\u003esubtree_control.\n\n    * cgroup-\u003echild_subsys_mask -\u003e cgroup-\u003esubtree_ss_mask\n    * cgroup_calc_child_subsys_mask() -\u003e cgroup_calc_subtree_ss_mask()\n    * cgroup_refresh_child_subsys_mask() -\u003e cgroup_refresh_subtree_ss_mask()\n\n    No functional changes.\n\n    Signed-off-by: Tejun Heo \u003ctj@kernel.org\u003e\n    Acked-by: Johannes Weiner \u003channes@cmpxchg.org\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 0317250e21ee50cb18dd1f86efddfa38f2953f9d\nAuthor: Tejun Heo \u003ctj@kernel.org\u003e\nDate:   Mon Feb 22 22:25:46 2016 -0500\n\n    Revert \"cgroup: add cgroup_subsys-\u003ecss_e_css_changed()\"\n\n    This reverts commit 56c807ba4e91f0980567b6a69de239677879b17f.\n\n    cgroup_subsys-\u003ecss_e_css_changed() was supposed to be used by cgroup\n    writeback support; however, the change to per-inode cgroup association\n    made it unnecessary and the callback doesn\u0027t have any user.  Remove\n    it.\n\n    Signed-off-by: Tejun Heo \u003ctj@kernel.org\u003e\n    Acked-by: Johannes Weiner \u003channes@cmpxchg.org\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 1b0104f0af0922a6de76ee0eaa2af1ef63ced0de\nAuthor: Tejun Heo \u003ctj@kernel.org\u003e\nDate:   Mon Feb 22 22:25:45 2016 -0500\n\n    cgroup: fix error return value of cgroup_addrm_files()\n\n    cgroup_addrm_files() incorrectly returned 0 after add failure.  Fix\n    it.\n\n    Signed-off-by: Tejun Heo \u003ctj@kernel.org\u003e\n    Acked-by: Johannes Weiner \u003channes@cmpxchg.org\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 959c3f17a0eedc34992cde0c4c7e3d3225d6d8b4\nAuthor: Tejun Heo \u003ctj@kernel.org\u003e\nDate:   Thu Feb 18 11:44:24 2016 -0500\n\n    cgroup: fix alloc_cgroup_ns() error handling in copy_cgroup_ns()\n\n    alloc_cgroup_ns() returns an ERR_PTR value on error but\n    copy_cgroup_ns() was checking for NULL for error.  Fix it.\n\n    Signed-off-by: Tejun Heo \u003ctj@kernel.org\u003e\n    Reported-by: Dan Carpenter \u003cdan.carpenter@oracle.com\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 6a2f54048873dfc28666c7c8c2e006f774eba02b\nAuthor: Serge Hallyn \u003cserge.hallyn@ubuntu.com\u003e\nDate:   Fri Jan 29 02:54:11 2016 -0600\n\n    Add FS_USERNS_FLAG to cgroup fs\n\n    allowing root in a non-init user namespace to mount it.  This should\n    now be safe, because\n\n    1. non-init-root cannot mount a previously unbound subsystem\n    2. the task doing the mount must be privileged with respect to the\n       user namespace owning the cgroup namespace\n    3. the mounted subsystem will have its current cgroup as the root dentry.\n       the permissions will be unchanged, so tasks will receive no new\n       privilege over the cgroups which they did not have on the original\n       mounts.\n\n    Signed-off-by: Serge Hallyn \u003cserge.hallyn@canonical.com\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 48a66647cab717b65b80e821475a3b89295db092\nAuthor: Serge Hallyn \u003cserge.hallyn@ubuntu.com\u003e\nDate:   Fri Jan 29 02:54:09 2016 -0600\n\n    cgroup: mount cgroupns-root when inside non-init cgroupns\n\n    This patch enables cgroup mounting inside userns when a process\n    as appropriate privileges. The cgroup filesystem mounted is\n    rooted at the cgroupns-root. Thus, in a container-setup, only\n    the hierarchy under the cgroupns-root is exposed inside the container.\n    This allows container management tools to run inside the containers\n    without depending on any global state.\n\n    Signed-off-by: Serge Hallyn \u003cserge.hallyn@canonical.com\u003e\n    Signed-off-by: Tejun Heo \u003ctj@kernel.org\u003e\n    Signed-off-by: Tejun Heo \u003ctj@kernel.org\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit d14c56b424aeac181246e5a2b7303b538d919bca\nAuthor: Aditya Kali \u003cadityakali@google.com\u003e\nDate:   Fri Jan 29 02:54:07 2016 -0600\n\n    cgroup: cgroup namespace setns support\n\n    setns on a cgroup namespace is allowed only if\n    task has CAP_SYS_ADMIN in its current user-namespace and\n    over the user-namespace associated with target cgroupns.\n    No implicit cgroup changes happen with attaching to another\n    cgroupns. It is expected that the somone moves the attaching\n    process under the target cgroupns-root.\n\n    Signed-off-by: Aditya Kali \u003cadityakali@google.com\u003e\n    Signed-off-by: Serge E. Hallyn \u003cserge.hallyn@canonical.com\u003e\n    Signed-off-by: Tejun Heo \u003ctj@kernel.org\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit f1c91cd3381c6f0100f6cca723e760373e74ee6e\nAuthor: Aditya Kali \u003cadityakali@google.com\u003e\nDate:   Fri Jan 29 02:54:06 2016 -0600\n\n    cgroup: introduce cgroup namespaces\n\n    Introduce the ability to create new cgroup namespace. The newly created\n    cgroup namespace remembers the cgroup of the process at the point\n    of creation of the cgroup namespace (referred as cgroupns-root).\n    The main purpose of cgroup namespace is to virtualize the contents\n    of /proc/self/cgroup file. Processes inside a cgroup namespace\n    are only able to see paths relative to their namespace root\n    (unless they are moved outside of their cgroupns-root, at which point\n     they will see a relative path from their cgroupns-root).\n    For a correctly setup container this enables container-tools\n    (like libcontainer, lxc, lmctfy, etc.) to create completely virtualized\n    containers without leaking system level cgroup hierarchy to the task.\n    This patch only implements the \u0027unshare\u0027 part of the cgroupns.\n\n    Signed-off-by: Aditya Kali \u003cadityakali@google.com\u003e\n    Signed-off-by: Serge Hallyn \u003cserge.hallyn@canonical.com\u003e\n    Signed-off-by: Tejun Heo \u003ctj@kernel.org\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n    Change-Id: I41c6f048567d7ab086467c6a230780c1858b315d\n\ncommit c09890431c1e95bb6a103d53dcb0784a73b68b04\nAuthor: Johannes Weiner \u003channes@cmpxchg.org\u003e\nDate:   Thu Feb 11 13:34:49 2016 -0500\n\n    cgroup: provide cgroup_nov1\u003d to disable controllers in v1 mounts\n\n    Testing cgroup2 can be painful with system software automatically\n    mounting and populating all cgroup controllers in v1 mode. Sometimes\n    they can be unmounted from rc.local, sometimes even that is too late.\n\n    Provide a commandline option to disable certain controllers in v1\n    mounts, so that they remain available for cgroup2 mounts.\n\n    Example use:\n\n    cgroup_no_v1\u003dmemory,cpu\n    cgroup_no_v1\u003dall\n\n    Disabling will be confirmed at boot-time as such:\n\n    [    0.013770] Disabling cpu control group subsystem in v1 mounts\n    [    0.016004] Disabling memory control group subsystem in v1 mounts\n\n    Signed-off-by: Johannes Weiner \u003channes@cmpxchg.org\u003e\n    Signed-off-by: Tejun Heo \u003ctj@kernel.org\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 08730efa99ba420805acf2dbaa5aa4f2be5238bd\nAuthor: Tejun Heo \u003ctj@kernel.org\u003e\nDate:   Tue Dec 29 14:53:56 2015 -0500\n\n    cgroup: demote subsystem init messages to KERN_DEBUG\n\n    These are noisy during boot and not all that interesting.\n\n    Signed-off-by: Tejun Heo \u003ctj@kernel.org\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit f6463c52a5107937c1b383bca79a2143f533b225\nAuthor: Rami Rosen \u003crami.rosen@intel.com\u003e\nDate:   Sat Jan 9 23:33:06 2016 +0200\n\n    cgroup: fix a typo.\n\n    This patch fixes a typo in a comment in cgroup.c.\n\n    Signed-off-by: Rami Rosen \u003crami.rosen@intel.com\u003e\n    Signed-off-by: Tejun Heo \u003ctj@kernel.org\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 0d1d0895cd450944c5b82914fcc35808b2224915\nAuthor: Tejun Heo \u003ctj@kernel.org\u003e\nDate:   Mon Dec 14 11:24:06 2015 -0500\n\n    net, cgroup: cgroup_sk_updat_lock was missing initializer\n\n    bd1060a1d671 (\"sock, cgroup: add sock-\u003esk_cgroup\") added global\n    spinlock cgroup_sk_update_lock but erroneously skipped initializer\n    leading to uninitialized spinlock warning.  Fix it by using\n    DEFINE_SPINLOCK().\n\n    Signed-off-by: Tejun Heo \u003ctj@kernel.org\u003e\n    Reported-by: Dexuan Cui \u003cdecui@microsoft.com\u003e\n    Fixes: bd1060a1d671 (\"sock, cgroup: add sock-\u003esk_cgroup\")\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 4daebc81aa70048462f8290069926d378651256e\nAuthor: Tejun Heo \u003ctj@kernel.org\u003e\nDate:   Mon Dec 7 17:38:53 2015 -0500\n\n    sock, cgroup: add sock-\u003esk_cgroup\n\n    In cgroup v1, dealing with cgroup membership was difficult because the\n    number of membership associations was unbound.  As a result, cgroup v1\n    grew several controllers whose primary purpose is either tagging\n    membership or pull in configuration knobs from other subsystems so\n    that cgroup membership test can be avoided.\n\n    net_cls and net_prio controllers are examples of the latter.  They\n    allow configuring network-specific attributes from cgroup side so that\n    network subsystem can avoid testing cgroup membership; unfortunately,\n    these are not only cumbersome but also problematic.\n\n    Both net_cls and net_prio aren\u0027t properly hierarchical.  Both inherit\n    configuration from the parent on creation but there\u0027s no interaction\n    afterwards.  An ancestor doesn\u0027t restrict the behavior in its subtree\n    in anyway and configuration changes aren\u0027t propagated downwards.\n    Especially when combined with cgroup delegation, this is problematic\n    because delegatees can mess up whatever network configuration\n    implemented at the system level.  net_prio would allow the delegatees\n    to set whatever priority value regardless of CAP_NET_ADMIN and net_cls\n    the same for classid.\n\n    While it is possible to solve these issues from controller side by\n    implementing hierarchical allowable ranges in both controllers, it\n    would involve quite a bit of complexity in the controllers and further\n    obfuscate network configuration as it becomes even more difficult to\n    tell what\u0027s actually being configured looking from the network side.\n    While not much can be done for v1 at this point, as membership\n    handling is sane on cgroup v2, it\u0027d be better to make cgroup matching\n    behave like other network matches and classifiers than introducing\n    further complications.\n\n    In preparation, this patch updates sock-\u003esk_cgrp_data handling so that\n    it points to the v2 cgroup that sock was created in until either\n    net_prio or net_cls is used.  Once either of the two is used,\n    sock-\u003esk_cgrp_data reverts to its previous role of carrying prioidx\n    and classid.  This is to avoid adding yet another cgroup related field\n    to struct sock.\n\n    As the mode switching can happen at most once per boot, the switching\n    mechanism is aimed at lowering hot path overhead.  It may leak a\n    finite, likely small, number of cgroup refs and report spurious\n    prioidx or classid on switching; however, dynamic updates of prioidx\n    and classid have always been racy and lossy - socks between creation\n    and fd installation are never updated, config changes don\u0027t update\n    existing sockets at all, and prioidx may index with dead and recycled\n    cgroup IDs.  Non-critical inaccuracies from small race windows won\u0027t\n    make any noticeable difference.\n\n    This patch doesn\u0027t make use of the pointer yet.  The following patch\n    will implement netfilter match for cgroup2 membership.\n\n    v2: Use sock_cgroup_data to avoid inflating struct sock w/ another\n        cgroup specific field.\n\n    v3: Add comments explaining why sock_data_prioidx() and\n        sock_data_classid() use different fallback values.\n\n    Signed-off-by: Tejun Heo \u003ctj@kernel.org\u003e\n    Cc: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Cc: Daniel Wagner \u003cdaniel.wagner@bmw-carit.de\u003e\n    CC: Neil Horman \u003cnhorman@tuxdriver.com\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit bcc4436d162b13051f4ef73b0386bb73dff412ef\nAuthor: Tejun Heo \u003ctj@kernel.org\u003e\nDate:   Mon Dec 7 17:38:52 2015 -0500\n\n    net: wrap sock-\u003esk_cgrp_prioidx and -\u003esk_classid inside a struct\n\n    Introduce sock-\u003esk_cgrp_data which is a struct sock_cgroup_data.\n    -\u003esk_cgroup_prioidx and -\u003esk_classid are moved into it.  The struct\n    and its accessors are defined in cgroup-defs.h.  This is to prepare\n    for overloading the fields with a cgroup pointer.\n\n    This patch mostly performs equivalent conversions but the followings\n    are noteworthy.\n\n    * Equality test before updating classid is removed from\n      sock_update_classid().  This shouldn\u0027t make any noticeable\n      difference and a similar test will be implemented on the helper side\n      later.\n\n    * sock_update_netprioidx() now takes struct sock_cgroup_data and can\n      be moved to netprio_cgroup.h without causing include dependency\n      loop.  Moved.\n\n    * The dummy version of sock_update_netprioidx() converted to a static\n      inline function while at it.\n\n    Signed-off-by: Tejun Heo \u003ctj@kernel.org\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 63b1c5c5c5a3ccf4a190577b5288ae6a581a186a\nAuthor: Tejun Heo \u003ctj@kernel.org\u003e\nDate:   Mon Dec 7 17:38:51 2015 -0500\n\n    netprio_cgroup: limit the maximum css-\u003eid to USHRT_MAX\n\n    netprio builds per-netdev contiguous priomap array which is indexed by\n    css-\u003eid.  The array is allocated using kzalloc() effectively limiting\n    the maximum ID supported to some thousand range.  This patch caps the\n    maximum supported css-\u003eid to USHRT_MAX which should be way above what\n    is actually useable.\n\n    This allows reducing sock-\u003esk_cgrp_prioidx to u16 from u32.  The freed\n    up part will be used to overload the cgroup related fields.\n    sock-\u003esk_cgrp_prioidx\u0027s position is swapped with sk_mark so that the\n    two cgroup related fields are adjacent.\n\n    Signed-off-by: Tejun Heo \u003ctj@kernel.org\u003e\n    Acked-by: Daniel Wagner \u003cdaniel.wagner@bmw-carit.de\u003e\n    Cc: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    CC: Neil Horman \u003cnhorman@tuxdriver.com\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit e98dcefa5b11c7d168adf8cb601a62a3d1e66450\nAuthor: Oleg Nesterov \u003coleg@redhat.com\u003e\nDate:   Thu Dec 3 10:24:08 2015 -0500\n\n    cgroup: kill cgrp_ss_priv[CGROUP_CANFORK_COUNT] and friends\n\n    Now that nobody use the \"priv\" arg passed to can_fork/cancel_fork/fork we can\n    kill CGROUP_CANFORK_COUNT/SUBSYS_TAG/etc and cgrp_ss_priv[] in copy_process().\n\n    Signed-off-by: Oleg Nesterov \u003coleg@redhat.com\u003e\n    Signed-off-by: Tejun Heo \u003ctj@kernel.org\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n    Change-Id: I153eb067c3378de42ccfd4bf114763032bb260ec\n\ncommit 283d51037e28a568451bd41fa5e90a3ce478cb0b\nAuthor: Tejun Heo \u003ctj@kernel.org\u003e\nDate:   Fri Nov 20 15:55:52 2015 -0500\n\n    cgroup: implement cgroup_get_from_path() and expose cgroup_put()\n\n    Implement cgroup_get_from_path() using kernfs_walk_and_get() which\n    obtains a default hierarchy cgroup from its path.  This will be used\n    to allow cgroup path based matching from outside cgroup proper -\n    e.g. networking and perf.\n\n    v2: Add EXPORT_SYMBOL_GPL(cgroup_get_from_path).\n\n    Signed-off-by: Tejun Heo \u003ctj@kernel.org\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit e397062e7a210236c7d184e029b587b81f57f5d2\nAuthor: Tejun Heo \u003ctj@kernel.org\u003e\nDate:   Fri Nov 20 15:55:52 2015 -0500\n\n    cgroup: record ancestor IDs and reimplement cgroup_is_descendant() using it\n\n    cgroup_is_descendant() currently walks up the hierarchy and compares\n    each ancestor to the cgroup in question.  While enough for cgroup core\n    usages, this can\u0027t be used in hot paths to test cgroup membership.\n    This patch adds cgroup-\u003eancestor_ids[] which records the IDs of all\n    ancestors including self and cgroup-\u003elevel for the nesting level.\n\n    This allows testing whether a given cgroup is a descendant of another\n    in three finite steps - testing whether the two belong to the same\n    hierarchy, whether the descendant candidate is at the same or a higher\n    level than the ancestor and comparing the recorded ancestor_id at the\n    matching level.  cgroup_is_descendant() is accordingly reimplmented\n    and made inline.\n\n    Signed-off-by: Tejun Heo \u003ctj@kernel.org\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit f4cdf9e7884cb2e332c9745116cbbd331e5c95c0\nAuthor: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\nDate:   Mon May 8 00:04:09 2017 +0200\n\n    bpf: don\u0027t let ldimm64 leak map addresses on unprivileged\n\n    [ Upstream commit 0d0e57697f162da4aa218b5feafe614fb666db07 ]\n\n    The patch fixes two things at once:\n\n    1) It checks the env-\u003eallow_ptr_leaks and only prints the map address to\n       the log if we have the privileges to do so, otherwise it just dumps 0\n       as we would when kptr_restrict is enabled on %pK. Given the latter is\n       off by default and not every distro sets it, I don\u0027t want to rely on\n       this, hence the 0 by default for unprivileged.\n\n    2) Printing of ldimm64 in the verifier log is currently broken in that\n       we don\u0027t print the full immediate, but only the 32 bit part of the\n       first insn part for ldimm64. Thus, fix this up as well; it\u0027s okay to\n       access, since we verified all ldimm64 earlier already (including just\n       constants) through replace_map_fd_with_map_ptr().\n\n    Fixes: 1be7f75d1668 (\"bpf: enable non-root eBPF programs\")\n    Fixes: cbd357008604 (\"bpf: verifier (add ability to receive verification log)\")\n    Reported-by: Jann Horn \u003cjannh@google.com\u003e\n    Signed-off-by: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Acked-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Greg Kroah-Hartman \u003cgregkh@linuxfoundation.org\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 94f835b9b83d3da72ee4773fe2014b1cf619dbe4\nAuthor: Yonghong Song \u003cyhs@fb.com\u003e\nDate:   Sat Apr 29 22:52:42 2017 -0700\n\n    bpf: enhance verifier to understand stack pointer arithmetic\n\n    [ Upstream commit 332270fdc8b6fba07d059a9ad44df9e1a2ad4529 ]\n\n    llvm 4.0 and above generates the code like below:\n    ....\n    440: (b7) r1 \u003d 15\n    441: (05) goto pc+73\n    515: (79) r6 \u003d *(u64 *)(r10 -152)\n    516: (bf) r7 \u003d r10\n    517: (07) r7 +\u003d -112\n    518: (bf) r2 \u003d r7\n    519: (0f) r2 +\u003d r1\n    520: (71) r1 \u003d *(u8 *)(r8 +0)\n    521: (73) *(u8 *)(r2 +45) \u003d r1\n    ....\n    and the verifier complains \"R2 invalid mem access \u0027inv\u0027\" for insn #521.\n    This is because verifier marks register r2 as unknown value after #519\n    where r2 is a stack pointer and r1 holds a constant value.\n\n    Teach verifier to recognize \"stack_ptr + imm\" and\n    \"stack_ptr + reg with const val\" as valid stack_ptr with new offset.\n\n    Signed-off-by: Yonghong Song \u003cyhs@fb.com\u003e\n    Acked-by: Martin KaFai Lau \u003ckafai@fb.com\u003e\n    Acked-by: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Signed-off-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Greg Kroah-Hartman \u003cgregkh@linuxfoundation.org\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 4252230cce4e92914affecd9cd2c50300db9fd8b\nAuthor: Alexei Starovoitov \u003cast@fb.com\u003e\nDate:   Fri Mar 24 15:57:33 2017 -0700\n\n    bpf: improve verifier packet range checks\n\n    [ Upstream commit b1977682a3858b5584ffea7cfb7bd863f68db18d ]\n\n    llvm can optimize the \u0027if (ptr \u003e data_end)\u0027 checks to be in the order\n    slightly different than the original C code which will confuse verifier.\n    Like:\n    if (ptr + 16 \u003e data_end)\n      return TC_ACT_SHOT;\n    // may be followed by\n    if (ptr + 14 \u003e data_end)\n      return TC_ACT_SHOT;\n    while llvm can see that \u0027ptr\u0027 is valid for all 16 bytes,\n    the verifier could not.\n    Fix verifier logic to account for such case and add a test.\n\n    Reported-by: Huapeng Zhou \u003chzhou@fb.com\u003e\n    Fixes: 969bf05eb3ce (\"bpf: direct packet access\")\n    Signed-off-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Acked-by: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Acked-by: Martin KaFai Lau \u003ckafai@fb.com\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Greg Kroah-Hartman \u003cgregkh@linuxfoundation.org\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit ef3e919d24610f2a2be4849d84af8f24b3d4efb7\nAuthor: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\nDate:   Sun Dec 18 01:52:59 2016 +0100\n\n    bpf: fix mark_reg_unknown_value for spilled regs on map value marking\n\n    [ Upstream commit 6760bf2ddde8ad64f8205a651223a93de3a35494 ]\n\n    Martin reported a verifier issue that hit the BUG_ON() for his\n    test case in the mark_reg_unknown_value() function:\n\n      [  202.861380] kernel BUG at kernel/bpf/verifier.c:467!\n      [...]\n      [  203.291109] Call Trace:\n      [  203.296501]  [\u003cffffffff811364d5\u003e] mark_map_reg+0x45/0x50\n      [  203.308225]  [\u003cffffffff81136558\u003e] mark_map_regs+0x78/0x90\n      [  203.320140]  [\u003cffffffff8113938d\u003e] do_check+0x226d/0x2c90\n      [  203.331865]  [\u003cffffffff8113a6ab\u003e] bpf_check+0x48b/0x780\n      [  203.343403]  [\u003cffffffff81134c8e\u003e] bpf_prog_load+0x27e/0x440\n      [  203.355705]  [\u003cffffffff8118a38f\u003e] ? handle_mm_fault+0x11af/0x1230\n      [  203.369158]  [\u003cffffffff812d8188\u003e] ? security_capable+0x48/0x60\n      [  203.382035]  [\u003cffffffff811351a4\u003e] SyS_bpf+0x124/0x960\n      [  203.393185]  [\u003cffffffff810515f6\u003e] ? __do_page_fault+0x276/0x490\n      [  203.406258]  [\u003cffffffff816db320\u003e] entry_SYSCALL_64_fastpath+0x13/0x94\n\n    This issue got uncovered after the fix in a08dd0da5307 (\"bpf: fix\n    regression on verifier pruning wrt map lookups\"). The reason why it\n    wasn\u0027t noticed before was, because as mentioned in a08dd0da5307,\n    mark_map_regs() was doing the id matching incorrectly based on the\n    uncached regs[regno].id. So, in the first loop, we walked all regs\n    and as soon as we found regno \u003d\u003d i, then this reg\u0027s id was cleared\n    when calling mark_reg_unknown_value() thus that every subsequent\n    register was probed against id of 0 (which, in combination with the\n    PTR_TO_MAP_VALUE_OR_NULL type is an invalid condition that no other\n    register state can hold), and therefore wasn\u0027t type transitioned such\n    as in the spilled register case for the second loop.\n\n    Now since that got fixed, it turned out that 57a09bf0a416 (\"bpf:\n    Detect identical PTR_TO_MAP_VALUE_OR_NULL registers\") used\n    mark_reg_unknown_value() incorrectly for the spilled regs, and thus\n    hitting the BUG_ON() in some cases due to regno \u003e\u003d MAX_BPF_REG.\n\n    Although spilled regs have the same type as the non-spilled regs\n    for the verifier state, that is, struct bpf_reg_state, they are\n    semantically different from the non-spilled regs. In other words,\n    there can be up to 64 (MAX_BPF_STACK / BPF_REG_SIZE) spilled regs\n    in the stack, for example, register R\u003cx\u003e could have been spilled by\n    the program to stack location X, Y, Z, and in mark_map_regs() we\n    need to scan these stack slots of type STACK_SPILL for potential\n    registers that we have to transition from PTR_TO_MAP_VALUE_OR_NULL.\n    Therefore, depending on the location, the spilled_regs regno can\n    be a lot higher than just MAX_BPF_REG\u0027s value since we operate on\n    stack instead. The reset in mark_reg_unknown_value() itself is\n    just fine, only that the BUG_ON() was inappropriate for this. Fix\n    it by making a __mark_reg_unknown_value() version that can be\n    called from mark_map_reg() generically; we know for the non-spilled\n    case that the regno is always \u003c MAX_BPF_REG anyway.\n\n    Fixes: 57a09bf0a416 (\"bpf: Detect identical PTR_TO_MAP_VALUE_OR_NULL registers\")\n    Reported-by: Martin KaFai Lau \u003ckafai@fb.com\u003e\n    Signed-off-by: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Acked-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Greg Kroah-Hartman \u003cgregkh@linuxfoundation.org\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 53024f8bfa94807c76a29820f2fcd0f015311f65\nAuthor: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\nDate:   Thu Dec 15 01:30:06 2016 +0100\n\n    bpf: fix regression on verifier pruning wrt map lookups\n\n    [ Upstream commit a08dd0da5307ba01295c8383923e51e7997c3576 ]\n\n    Commit 57a09bf0a416 (\"bpf: Detect identical PTR_TO_MAP_VALUE_OR_NULL\n    registers\") introduced a regression where existing programs stopped\n    loading due to reaching the verifier\u0027s maximum complexity limit,\n    whereas prior to this commit they were loading just fine; the affected\n    program has roughly 2k instructions.\n\n    What was found is that state pruning couldn\u0027t be performed effectively\n    anymore due to mismatches of the verifier\u0027s register state, in particular\n    in the id tracking. It doesn\u0027t mean that 57a09bf0a416 is incorrect per\n    se, but rather that verifier needs to perform a lot more work for the\n    same program with regards to involved map lookups.\n\n    Since commit 57a09bf0a416 is only about tracking registers with type\n    PTR_TO_MAP_VALUE_OR_NULL, the id is only needed to follow registers\n    until they are promoted through pattern matching with a NULL check to\n    either PTR_TO_MAP_VALUE or UNKNOWN_VALUE type. After that point, the\n    id becomes irrelevant for the transitioned types.\n\n    For UNKNOWN_VALUE, id is already reset to 0 via mark_reg_unknown_value(),\n    but not so for PTR_TO_MAP_VALUE where id is becoming stale. It\u0027s even\n    transferred further into other types that don\u0027t make use of it. Among\n    others, one example is where UNKNOWN_VALUE is set on function call\n    return with RET_INTEGER return type.\n\n    states_equal() will then fall through the memcmp() on register state;\n    note that the second memcmp() uses offsetofend(), so the id is part of\n    that since d2a4dd37f6b4 (\"bpf: fix state equivalence\"). But the bisect\n    pointed already to 57a09bf0a416, where we really reach beyond complexity\n    limit. What I found was that states_equal() often failed in this\n    case due to id mismatches in spilled regs with registers in type\n    PTR_TO_MAP_VALUE. Unlike non-spilled regs, spilled regs just perform\n    a memcmp() on their reg state and don\u0027t have any other optimizations\n    in place, therefore also id was relevant in this case for making a\n    pruning decision.\n\n    We can safely reset id to 0 as well when converting to PTR_TO_MAP_VALUE.\n    For the affected program, it resulted in a ~17 fold reduction of\n    complexity and let the program load fine again. Selftest suite also\n    runs fine. The only other place where env-\u003eid_gen is used currently is\n    through direct packet access, but for these cases id is long living, thus\n    a different scenario.\n\n    Also, the current logic in mark_map_regs() is not fully correct when\n    marking NULL branch with UNKNOWN_VALUE. We need to cache the destination\n    reg\u0027s id in any case. Otherwise, once we marked that reg as UNKNOWN_VALUE,\n    it\u0027s id is reset and any subsequent registers that hold the original id\n    and are of type PTR_TO_MAP_VALUE_OR_NULL won\u0027t be marked UNKNOWN_VALUE\n    anymore, since mark_map_reg() reuses the uncached regs[regno].id that\n    was just overridden. Note, we don\u0027t need to cache it outside of\n    mark_map_regs(), since it\u0027s called once on this_branch and the other\n    time on other_branch, which are both two independent verifier states.\n    A test case for this is added here, too.\n\n    Fixes: 57a09bf0a416 (\"bpf: Detect identical PTR_TO_MAP_VALUE_OR_NULL registers\")\n    Signed-off-by: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Acked-by: Thomas Graf \u003ctgraf@suug.ch\u003e\n    Acked-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Greg Kroah-Hartman \u003cgregkh@linuxfoundation.org\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 4f4d7ac9ac2783764c9ec16153eab792abec40a0\nAuthor: Alexei Starovoitov \u003cast@fb.com\u003e\nDate:   Wed Dec 7 10:57:59 2016 -0800\n\n    bpf: fix state equivalence\n\n    [ Upstream commit d2a4dd37f6b41fbcad76efbf63124eb3126c66fe ]\n\n    Commmits 57a09bf0a416 (\"bpf: Detect identical PTR_TO_MAP_VALUE_OR_NULL registers\")\n    and 484611357c19 (\"bpf: allow access into map value arrays\") by themselves\n    are correct, but in combination they make state equivalence ignore \u0027id\u0027 field\n    of the register state which can lead to accepting invalid program.\n\n    Fixes: 57a09bf0a416 (\"bpf: Detect identical PTR_TO_MAP_VALUE_OR_NULL registers\")\n    Fixes: 484611357c19 (\"bpf: allow access into map value arrays\")\n    Signed-off-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Acked-by: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Acked-by: Thomas Graf \u003ctgraf@suug.ch\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Greg Kroah-Hartman \u003cgregkh@linuxfoundation.org\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit ad0e8f5401a20f2fe5da005f1aa9573a0e001f84\nAuthor: Thomas Graf \u003ctgraf@suug.ch\u003e\nDate:   Tue Oct 18 19:51:19 2016 +0200\n\n    bpf: Detect identical PTR_TO_MAP_VALUE_OR_NULL registers\n\n    [ Upstream commit 57a09bf0a416700676e77102c28f9cfcb48267e0 ]\n\n    A BPF program is required to check the return register of a\n    map_elem_lookup() call before accessing memory. The verifier keeps\n    track of this by converting the type of the result register from\n    PTR_TO_MAP_VALUE_OR_NULL to PTR_TO_MAP_VALUE after a conditional\n    jump ensures safety. This check is currently exclusively performed\n    for the result register 0.\n\n    In the event the compiler reorders instructions, BPF_MOV64_REG\n    instructions may be moved before the conditional jump which causes\n    them to keep their type PTR_TO_MAP_VALUE_OR_NULL to which the\n    verifier objects when the register is accessed:\n\n    0: (b7) r1 \u003d 10\n    1: (7b) *(u64 *)(r10 -8) \u003d r1\n    2: (bf) r2 \u003d r10\n    3: (07) r2 +\u003d -8\n    4: (18) r1 \u003d 0x59c00000\n    6: (85) call 1\n    7: (bf) r4 \u003d r0\n    8: (15) if r0 \u003d\u003d 0x0 goto pc+1\n     R0\u003dmap_value(ks\u003d8,vs\u003d8) R4\u003dmap_value_or_null(ks\u003d8,vs\u003d8) R10\u003dfp\n    9: (7a) *(u64 *)(r4 +0) \u003d 0\n    R4 invalid mem access \u0027map_value_or_null\u0027\n\n    This commit extends the verifier to keep track of all identical\n    PTR_TO_MAP_VALUE_OR_NULL registers after a map_elem_lookup() by\n    assigning them an ID and then marking them all when the conditional\n    jump is observed.\n\n    Signed-off-by: Thomas Graf \u003ctgraf@suug.ch\u003e\n    Reviewed-by: Josef Bacik \u003cjbacik@fb.com\u003e\n    Acked-by: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Acked-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Greg Kroah-Hartman \u003cgregkh@linuxfoundation.org\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 172f58ef19644fcd8ba36d8ce7c0d83d0acb9d89\nAuthor: Josef Bacik \u003cjbacik@fb.com\u003e\nDate:   Tue Nov 29 12:27:09 2016 -0500\n\n    bpf: fix states equal logic for varlen access\n\n    If we have a branch that looks something like this\n\n    int foo \u003d map-\u003evalue;\n    if (condition) {\n      foo +\u003d blah;\n    } else {\n      foo \u003d bar;\n    }\n    map-\u003earray[foo] \u003d baz;\n\n    We will incorrectly assume that the !condition branch is equal to the condition\n    branch as the register for foo will be UNKNOWN_VALUE in both cases.  We need to\n    adjust this logic to only do this if we didn\u0027t do a varlen access after we\n    processed the !condition branch, otherwise we have different ranges and need to\n    check the other branch as well.\n\n    Fixes: 484611357c19 (\"bpf: allow access into map value arrays\")\n    Reported-by: Jann Horn \u003cjannh@google.com\u003e\n    Signed-off-by: Josef Bacik \u003cjbacik@fb.com\u003e\n    Acked-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Acked-by: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit c630a7b76841d03f46127abc5bf40200946dc5f9\nAuthor: Josef Bacik \u003cjbacik@fb.com\u003e\nDate:   Mon Nov 14 15:45:36 2016 -0500\n\n    bpf: fix range arithmetic for bpf map access\n\n    I made some invalid assumptions with BPF_AND and BPF_MOD that could result in\n    invalid accesses to bpf map entries.  Fix this up by doing a few things\n\n    1) Kill BPF_MOD support.  This doesn\u0027t actually get used by the compiler in real\n    life and just adds extra complexity.\n\n    2) Fix the logic for BPF_AND, don\u0027t allow AND of negative numbers and set the\n    minimum value to 0 for positive AND\u0027s.\n\n    3) Don\u0027t do operations on the ranges if they are set to the limits, as they are\n    by definition undefined, and allowing arithmetic operations on those values\n    could make them appear valid when they really aren\u0027t.\n\n    This fixes the testcase provided by Jann as well as a few other theoretical\n    problems.\n\n    Reported-by: Jann Horn \u003cjannh@google.com\u003e\n    Signed-off-by: Josef Bacik \u003cjbacik@fb.com\u003e\n    Acked-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 7240f24fe3da999d8655484d6da877fadfc6b5bf\nAuthor: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\nDate:   Fri Nov 4 00:01:19 2016 +0100\n\n    bpf: fix htab map destruction when extra reserve is in use\n\n    Commit a6ed3ea65d98 (\"bpf: restore behavior of bpf_map_update_elem\")\n    added an extra per-cpu reserve to the hash table map to restore old\n    behaviour from pre prealloc times. When non-prealloc is in use for a\n    map, then problem is that once a hash table extra element has been\n    linked into the hash-table, and the hash table is destroyed due to\n    refcount dropping to zero, then htab_map_free() -\u003e delete_all_elements()\n    will walk the whole hash table and drop all elements via htab_elem_free().\n    The problem is that the element from the extra reserve is first fed\n    to the wrong backend allocator and eventually freed twice.\n\n    Fixes: a6ed3ea65d98 (\"bpf: restore behavior of bpf_map_update_elem\")\n    Reported-by: Dmitry Vyukov \u003cdvyukov@google.com\u003e\n    Signed-off-by: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Acked-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 0057b23ea73ebd20042d15d75b96d95ab631fabc\nAuthor: Josef Bacik \u003cjbacik@fb.com\u003e\nDate:   Wed Sep 28 10:54:32 2016 -0400\n\n    bpf: allow access into map value arrays\n\n    Suppose you have a map array value that is something like this\n\n    struct foo {\n    \tunsigned iter;\n    \tint array[SOME_CONSTANT];\n    };\n\n    You can easily insert this into an array, but you cannot modify the contents of\n    foo-\u003earray[] after the fact.  This is because we have no way to verify we won\u0027t\n    go off the end of the array at verification time.  This patch provides a start\n    for this work.  We accomplish this by keeping track of a minimum and maximum\n    value a register could be while we\u0027re checking the code.  Then at the time we\n    try to do an access into a MAP_VALUE we verify that the maximum offset into that\n    region is a valid access into that memory region.  So in practice, code such as\n    this\n\n    unsigned index \u003d 0;\n\n    if (foo-\u003eiter \u003e\u003d SOME_CONSTANT)\n    \tfoo-\u003eiter \u003d index;\n    else\n    \tindex \u003d foo-\u003eiter++;\n    foo-\u003earray[index] \u003d bar;\n\n    would be allowed, as we can verify that index will always be between 0 and\n    SOME_CONSTANT-1.  If you wish to use signed values you\u0027ll have to have an extra\n    check to make sure the index isn\u0027t less than 0, or do something like index %\u003d\n    SOME_CONSTANT.\n\n    Signed-off-by: Josef Bacik \u003cjbacik@fb.com\u003e\n    Acked-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 60cf4edb6bc35c4766a3a498c4109534bf8f2ba0\nAuthor: Shaohua Li \u003cshli@fb.com\u003e\nDate:   Tue Sep 27 08:42:41 2016 -0700\n\n    bpf: clean up put_cpu_var usage\n\n    put_cpu_var takes the percpu data, not the data returned from\n    get_cpu_var.\n\n    This doesn\u0027t change the behavior.\n\n    Cc: Tejun Heo \u003ctj@kernel.org\u003e\n    Cc: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Signed-off-by: Shaohua Li \u003cshli@fb.com\u003e\n    Acked-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Acked-by: Tejun Heo \u003ctj@kernel.org\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit f340f5513e703e114493ac95ecebbf47d7881c6c\nAuthor: Mickaël Salaün \u003cmic@digikod.net\u003e\nDate:   Sat Sep 24 20:01:50 2016 +0200\n\n    bpf: Set register type according to is_valid_access()\n\n    This prevent future potential pointer leaks when an unprivileged eBPF\n    program will read a pointer value from its context. Even if\n    is_valid_access() returns a pointer type, the eBPF verifier replace it\n    with UNKNOWN_VALUE. The register value that contains a kernel address is\n    then allowed to leak. Moreover, this fix allows unprivileged eBPF\n    programs to use functions with (legitimate) pointer arguments.\n\n    Not an issue currently since reg_type is only set for PTR_TO_PACKET or\n    PTR_TO_PACKET_END in XDP and TC programs that can only be loaded as\n    privileged. For now, the only unprivileged eBPF program allowed is for\n    socket filtering and all the types from its context are UNKNOWN_VALUE.\n    However, this fix is important for future unprivileged eBPF programs\n    which could use pointers in their context.\n\n    Signed-off-by: Mickaël Salaün \u003cmic@digikod.net\u003e\n    Cc: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Cc: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Acked-by: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Acked-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 3b445710a196d07dc196266bce694fc9ae83004b\nAuthor: Jakub Kicinski \u003cjakub.kicinski@netronome.com\u003e\nDate:   Wed Sep 21 11:43:59 2016 +0100\n\n    bpf: recognize 64bit immediate loads as consts\n\n    When running as parser interpret BPF_LD | BPF_IMM | BPF_DW\n    instructions as loading CONST_IMM with the value stored\n    in imm.  The verifier will continue not recognizing those\n    due to concerns about search space/program complexity\n    increase.\n\n    Signed-off-by: Jakub Kicinski \u003cjakub.kicinski@netronome.com\u003e\n    Acked-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Acked-by: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 9e6fd59da8035b39a770cb1b9508bce785898f38\nAuthor: Jakub Kicinski \u003cjakub.kicinski@netronome.com\u003e\nDate:   Wed Sep 21 11:43:58 2016 +0100\n\n    bpf: enable non-core use of the verfier\n\n    Advanced JIT compilers and translators may want to use\n    eBPF verifier as a base for parsers or to perform custom\n    checks and validations.\n\n    Add ability for external users to invoke the verifier\n    and provide callbacks to be invoked for every intruction\n    checked.  For now only add most basic callback for\n    per-instruction pre-interpretation checks is added.  More\n    advanced users may also like to have per-instruction post\n    callback and state comparison callback.\n\n    Signed-off-by: Jakub Kicinski \u003cjakub.kicinski@netronome.com\u003e\n    Acked-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 2f728ad59eedee6a327dd59d21210f285dd9a0fe\nAuthor: Jakub Kicinski \u003cjakub.kicinski@netronome.com\u003e\nDate:   Wed Sep 21 11:43:57 2016 +0100\n\n    bpf: expose internal verfier structures\n\n    Move verifier\u0027s internal structures to a header file and\n    prefix their names with bpf_ to avoid potential namespace\n    conflicts.  Those structures will soon be used by external\n    analyzers.\n\n    Signed-off-by: Jakub Kicinski \u003cjakub.kicinski@netronome.com\u003e\n    Acked-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Acked-by: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 22e7afab4576f616726444590ee2e5005eb49d3f\nAuthor: Jakub Kicinski \u003cjakub.kicinski@netronome.com\u003e\nDate:   Wed Sep 21 11:43:56 2016 +0100\n\n    bpf: don\u0027t (ab)use instructions to store state\n\n    Storing state in reserved fields of instructions makes\n    it impossible to run verifier on programs already\n    marked as read-only. Allocate and use an array of\n    per-instruction state instead.\n\n    While touching the error path rename and move existing\n    jump target.\n\n    Suggested-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Signed-off-by: Jakub Kicinski \u003cjakub.kicinski@netronome.com\u003e\n    Acked-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Acked-by: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 30d07135dc717446771544bf52ae028e8937d226\nAuthor: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\nDate:   Tue Sep 20 00:26:13 2016 +0200\n\n    bpf: direct packet write and access for helpers for clsact progs\n\n    This work implements direct packet access for helpers and direct packet\n    write in a similar fashion as already available for XDP types via commits\n    4acf6c0b84c9 (\"bpf: enable direct packet data write for xdp progs\") and\n    6841de8b0d03 (\"bpf: allow helpers access the packet directly\"), and as a\n    complementary feature to the already available direct packet read for tc\n    (cls/act) programs.\n\n    For enabling this, we need to introduce two helpers, bpf_skb_pull_data()\n    and bpf_csum_update(). The first is generally needed for both, read and\n    write, because they would otherwise only be limited to the current linear\n    skb head. Usually, when the data_end test fails, programs just bail out,\n    or, in the direct read case, use bpf_skb_load_bytes() as an alternative\n    to overcome this limitation. If such data sits in non-linear parts, we\n    can just pull them in once with the new helper, retest and eventually\n    access them.\n\n    At the same time, this also makes sure the skb is uncloned, which is, of\n    course, a necessary condition for direct write. As this needs to be an\n    invariant for the write part only, the verifier detects writes and adds\n    a prologue that is calling bpf_skb_pull_data() to effectively unclone the\n    skb from the very beginning in case it is indeed cloned. The heuristic\n    makes use of a similar trick that was done in 233577a22089 (\"net: filter:\n    constify detection of pkt_type_offset\"). This comes at zero cost for other\n    programs that do not use the direct write feature. Should a program use\n    this feature only sparsely and has read access for the most parts with,\n    for example, drop return codes, then such write action can be delegated\n    to a tail called program for mitigating this cost of potential uncloning\n    to a late point in time where it would have been paid similarly with the\n    bpf_skb_store_bytes() as well. Advantage of direct write is that the\n    writes are inlined whereas the helper cannot make any length assumptions\n    and thus needs to generate a call to memcpy() also for small sizes, as well\n    as cost of helper call itself with sanity checks are avoided. Plus, when\n    direct read is already used, we don\u0027t need to cache or perform rechecks\n    on the data boundaries (due to verifier invalidating previous checks for\n    helpers that change skb-\u003edata), so more complex programs using rewrites\n    can benefit from switching to direct read plus write.\n\n    For direct packet access to helpers, we save the otherwise needed copy into\n    a temp struct sitting on stack memory when use-case allows. Both facilities\n    are enabled via may_access_direct_pkt_data() in verifier. For now, we limit\n    this to map helpers and csum_diff, and can successively enable other helpers\n    where we find it makes sense. Helpers that definitely cannot be allowed for\n    this are those part of bpf_helper_changes_skb_data() since they can change\n    underlying data, and those that write into memory as this could happen for\n    packet typed args when still cloned. bpf_csum_update() helper accommodates\n    for the fact that we need to fixup checksum_complete when using direct write\n    instead of bpf_skb_store_bytes(), meaning the programs can use available\n    helpers like bpf_csum_diff(), and implement csum_add(), csum_sub(),\n    csum_block_add(), csum_block_sub() equivalents in eBPF together with the\n    new helper. A usage example will be provided for iproute2\u0027s examples/bpf/\n    directory.\n\n    Signed-off-by: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Acked-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 5727cfee236ac7c9c761347a221af1cb0901875e\nAuthor: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\nDate:   Tue Sep 20 00:26:12 2016 +0200\n\n    bpf, verifier: enforce larger zero range for pkt on overloading stack buffs\n\n    Current contract for the following two helper argument types is:\n\n      * ARG_CONST_STACK_SIZE: passed argument pair must be (ptr, \u003e0).\n      * ARG_CONST_STACK_SIZE_OR_ZERO: passed argument pair can be either\n        (NULL, 0) or (ptr, \u003e0).\n\n    With 6841de8b0d03 (\"bpf: allow helpers access the packet directly\"), we can\n    pass also raw packet data to helpers, so depending on the argument type\n    being PTR_TO_PACKET, we now either assert memory via check_packet_access()\n    or check_stack_boundary(). As a result, the tests in check_packet_access()\n    currently allow more than intended with regards to reg-\u003eimm.\n\n    Back in 969bf05eb3ce (\"bpf: direct packet access\"), check_packet_access()\n    was fine to ignore size argument since in check_mem_access() size was\n    bpf_size_to_bytes() derived and prior to the call to check_packet_access()\n    guaranteed to be larger than zero.\n\n    However, for the above two argument types, it currently means, we can have\n    a \u003c\u003d 0 size and thus breaking current guarantees for helpers. Enforce a\n    check for size \u003c\u003d 0 and bail out if so.\n\n    check_stack_boundary() doesn\u0027t have such an issue since it already tests\n    for access_size \u003c\u003d 0 and bails out, resp. access_size \u003d\u003d 0 in case of NULL\n    pointer passed when allowed.\n\n    Fixes: 6841de8b0d03 (\"bpf: allow helpers access the packet directly\")\n    Signed-off-by: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Acked-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 3ecba0c15d9a3b68f5609e8819a8f2326698438b\nAuthor: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\nDate:   Fri Sep 9 02:45:31 2016 +0200\n\n    bpf: add BPF_CALL_x macros for declaring helpers\n\n    This work adds BPF_CALL_\u003cn\u003e() macros and converts all the eBPF helper functions\n    to use them, in a similar fashion like we do with SYSCALL_DEFINE\u003cn\u003e() macros\n    that are used today. Motivation for this is to hide all the register handling\n    and all necessary casts from the user, so that it is done automatically in the\n    background when adding a BPF_CALL_\u003cn\u003e() call.\n\n    This makes current helpers easier to review, eases to write future helpers,\n    avoids getting the casting mess wrong, and allows for extending all helpers at\n    once (f.e. build time checks, etc). It also helps detecting more easily in\n    code reviews that unused registers are not instrumented in the code by accident,\n    breaking compatibility with existing programs.\n\n    BPF_CALL_\u003cn\u003e() internals are quite similar to SYSCALL_DEFINE\u003cn\u003e() ones with some\n    fundamental differences, for example, for generating the actual helper function\n    that carries all u64 regs, we need to fill unused regs, so that we always end up\n    with 5 u64 regs as an argument.\n\n    I reviewed several 0-5 generated BPF_CALL_\u003cn\u003e() variants of the .i results and\n    they look all as expected. No sparse issue spotted. We let this also sit for a\n    few days with Fengguang\u0027s kbuild test robot, and there were no issues seen. On\n    s390, it barked on the \"uses dynamic stack allocation\" notice, which is an old\n    one from bpf_perf_event_output{,_tp}() reappearing here due to the conversion\n    to the call wrapper, just telling that the perf raw record/frag sits on stack\n    (gcc with s390\u0027s -mwarn-dynamicstack), but that\u0027s all. Did various runtime tests\n    and they were fine as well. All eBPF helpers are now converted to use these\n    macros, getting rid of a good chunk of all the raw castings.\n\n    Signed-off-by: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Acked-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 7838443f5d6207bee662afa8d1d4050aa78b5dd2\nAuthor: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\nDate:   Fri Aug 5 00:11:13 2016 +0200\n\n    bpf: fix checksum for vlan push/pop helper\n\n    When having skbs on ingress with CHECKSUM_COMPLETE, tc BPF programs don\u0027t\n    push rcsum of mac header back in and after BPF run back pull out again as\n    opposed to some other subsystems (ovs, for example).\n\n    For cases like q-in-q, meaning when a vlan tag for offloading is already\n    present and we\u0027re about to push another one, then skb_vlan_push() pushes the\n    inner one into the skb, increasing mac header and skb_postpush_rcsum()\u0027ing\n    the 4 bytes vlan header diff. Likewise, for the reverse operation in\n    skb_vlan_pop() for the case where vlan header needs to be pulled out of the\n    skb, we\u0027re decreasing the mac header and skb_postpull_rcsum()\u0027ing the 4 bytes\n    rcsum of the vlan header that was removed.\n\n    However mangling the rcsum here will lead to hw csum failure for BPF case,\n    since we\u0027re pulling or pushing data that was not part of the current rcsum.\n    Changing tc BPF programs in general to push/pull rcsum around BPF_PROG_RUN()\n    is also not really an option since current behaviour is ABI by now, but apart\n    from that would also mean to do quite a bit of useless work in the sense that\n    usually 12 bytes need to be rcsum pushed/pulled also when we don\u0027t need to\n    touch this vlan related corner case. One way to fix it would be to push the\n    necessary rcsum fixup down into vlan helpers that are (mostly) slow-path\n    anyway.\n\n    Fixes: 4e10df9a60d9 (\"bpf: introduce bpf_skb_vlan_push/pop() helpers\")\n    Signed-off-by: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Acked-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 430758126c5690f82ea177e764f5fb3129bdffc7\nAuthor: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\nDate:   Fri Aug 5 00:11:12 2016 +0200\n\n    bpf: fix checksum fixups on bpf_skb_store_bytes\n\n    bpf_skb_store_bytes() invocations above L2 header need BPF_F_RECOMPUTE_CSUM\n    flag for updates, so that CHECKSUM_COMPLETE will be fixed up along the way.\n    Where we ran into an issue with bpf_skb_store_bytes() is when we did a\n    single-byte update on the IPv6 hoplimit despite using BPF_F_RECOMPUTE_CSUM\n    flag; simple ping via ICMPv6 triggered a hw csum failure as a result. The\n    underlying issue has been tracked down to a buffer alignment issue.\n\n    Meaning, that csum_partial() computations via skb_postpull_rcsum() and\n    skb_postpush_rcsum() pair invoked had a wrong result since they operated on\n    an odd address for the hoplimit, while other computations were done on an\n    even address. This mix doesn\u0027t work as-is with skb_postpull_rcsum(),\n    skb_postpush_rcsum() pair as it always expects at least half-word alignment\n    of input buffers, which is normally the case. Thus, instead of these helpers\n    using csum_sub() and (implicitly) csum_add(), we need to use csum_block_sub(),\n    csum_block_add(), respectively. For unaligned offsets, they rotate the sum\n    to align it to a half-word boundary again, otherwise they work the same as\n    csum_sub() and csum_add().\n\n    Adding __skb_postpull_rcsum(), __skb_postpush_rcsum() variants that take the\n    offset as an input and adapting bpf_skb_store_bytes() to them fixes the hw\n    csum failures again. The skb_postpull_rcsum(), skb_postpush_rcsum() helpers\n    use a 0 constant for offset so that the compiler optimizes the offset \u0026 1\n    test away and generates the same code as with csum_sub()/_add().\n\n    Fixes: 608cd71a9c7c (\"tc: bpf: generalize pedit action\")\n    Signed-off-by: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Acked-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit b56fb8850be31217a4cc23c557994a964f3e3640\nAuthor: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\nDate:   Fri Aug 5 00:11:11 2016 +0200\n\n    bpf: also call skb_postpush_rcsum on xmit occasions\n\n    Follow-up to commit f8ffad69c9f8 (\"bpf: add skb_postpush_rcsum and fix\n    dev_forward_skb occasions\") to fix an issue for dev_queue_xmit() redirect\n    locations which need CHECKSUM_COMPLETE fixups on ingress.\n\n    For the same reasons as described in f8ffad69c9f8 already, we of course\n    also need this here, since dev_queue_xmit() on a veth device will let us\n    end up in the dev_forward_skb() helper again to cross namespaces.\n\n    Latter then calls into skb_postpull_rcsum() to pull out L2 header, so\n    that netif_rx_internal() sees CHECKSUM_COMPLETE as it is expected. That\n    is, CHECKSUM_COMPLETE on ingress covering L2 _payload_, not L2 headers.\n\n    Also here we have to address bpf_redirect() and bpf_clone_redirect().\n\n    Fixes: 3896d655f4d4 (\"bpf: introduce bpf_clone_redirect() helper\")\n    Fixes: 27b29f63058d (\"bpf: add bpf_redirect() helper\")\n    Signed-off-by: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Acked-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 936b09e8df85b8193f0a6e4577323d0e14d19419\nAuthor: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\nDate:   Fri Jun 10 21:19:06 2016 +0200\n\n    bpf: enforce recursion limit on redirects\n\n    Respect the stack\u0027s xmit_recursion limit for calls into dev_queue_xmit().\n    Currently, they are not handeled by the limiter when attached to clsact\u0027s\n    egress parent, for example, and a buggy program redirecting it to the\n    same device again could run into stack overflow eventually. It would be\n    good if we could notify an admin to give him a chance to react. We reuse\n    xmit_recursion instead of having one private to eBPF, so that the stack\u0027s\n    current recursion depth will be taken into account as well. Follow-up to\n    commit 3896d655f4d4 (\"bpf: introduce bpf_clone_redirect() helper\") and\n    27b29f63058d (\"bpf: add bpf_redirect() helper\").\n\n    Signed-off-by: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Acked-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit f401a2efb35493d5ac66c21c4a14fdfd9da56d9e\nAuthor: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\nDate:   Fri Sep 9 02:45:30 2016 +0200\n\n    bpf: add own ctx rewriter on ifindex for clsact progs\n\n    When fetching ifindex, we don\u0027t need to test dev for being NULL since\n    we\u0027re always guaranteed to have a valid dev for clsact programs. Thus,\n    avoid this test in fast path.\n\n    Signed-off-by: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Acked-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 3862495597f4a9de172a87b01c2ebeb4fbb64d05\nAuthor: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\nDate:   Thu Jul 14 18:08:04 2016 +0200\n\n    bpf, perf: split bpf_perf_event_output\n\n    Split the bpf_perf_event_output() helper as a preparation into\n    two parts. The new bpf_perf_event_output() will prepare the raw\n    record itself and test for unknown flags from BPF trace context,\n    where the __bpf_perf_event_output() does the core work. The\n    latter will be reused later on from bpf_event_output() directly.\n\n    Signed-off-by: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Acked-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit ec650fbdacc972f3ec5c570c72a804ccdf2d46bd\nAuthor: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\nDate:   Fri Sep 9 02:45:29 2016 +0200\n\n    bpf: add BPF_SIZEOF and BPF_FIELD_SIZEOF macros\n\n    Add BPF_SIZEOF() and BPF_FIELD_SIZEOF() macros to improve the code a bit\n    which otherwise often result in overly long bytes_to_bpf_size(sizeof())\n    and bytes_to_bpf_size(FIELD_SIZEOF()) lines. So place them into a macro\n    helper instead. Moreover, we currently have a BUILD_BUG_ON(BPF_FIELD_SIZEOF())\n    check in convert_bpf_extensions(), but we should rather make that generic\n    as well and add a BUILD_BUG_ON() test in all BPF_SIZEOF()/BPF_FIELD_SIZEOF()\n    users to detect any rewriter size issues at compile time. Note, there are\n    currently none, but we want to assert that it stays this way.\n\n    Signed-off-by: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Acked-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit f9bf2593b3f0dcd29447d2a9b4ec8f26e8efb8f6\nAuthor: Alexei Starovoitov \u003cast@fb.com\u003e\nDate:   Thu Sep 1 18:37:22 2016 -0700\n\n    bpf: introduce BPF_PROG_TYPE_PERF_EVENT program type\n\n    Introduce BPF_PROG_TYPE_PERF_EVENT programs that can be attached to\n    HW and SW perf events (PERF_TYPE_HARDWARE and PERF_TYPE_SOFTWARE\n    correspondingly in uapi/linux/perf_event.h)\n\n    The program visible context meta structure is\n    struct bpf_perf_event_data {\n        struct pt_regs regs;\n         __u64 sample_period;\n    };\n    which is accessible directly from the program:\n    int bpf_prog(struct bpf_perf_event_data *ctx)\n    {\n      ... ctx-\u003esample_period ...\n      ... ctx-\u003eregs.ip ...\n    }\n\n    The bpf verifier rewrites the accesses into kernel internal\n    struct bpf_perf_event_data_kern which allows changing\n    struct perf_sample_data without affecting bpf programs.\n    New fields can be added to the end of struct bpf_perf_event_data\n    in the future.\n\n    Signed-off-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Acked-by: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 57219c7eeaf3348a87a3bae0647510002dbd976d\nAuthor: Alexei Starovoitov \u003cast@fb.com\u003e\nDate:   Thu Aug 11 18:17:18 2016 -0700\n\n    bpf: allow bpf_get_prandom_u32() to be used in tracing\n\n    bpf_get_prandom_u32() was initially introduced for socket filters\n    and later requested numberous times to be added to tracing bpf programs\n    for the same reason as in socket filters: to be able to randomly\n    select incoming events.\n\n    Signed-off-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Acked-by: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 12892369eecbbe9b3cf6418f907b1d0550e322bc\nAuthor: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\nDate:   Fri Sep 9 02:45:28 2016 +0200\n\n    bpf: minor cleanups in helpers\n\n    Some minor misc cleanups, f.e. use sizeof(__u32) instead of hardcoding\n    and in __bpf_skb_max_len(), I missed that we always have skb-\u003edev valid\n    anyway, so we can drop the unneeded test for dev; also few more other\n    misc bits addressed here.\n\n    Signed-off-by: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Acked-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit a5da0516449f6bf024a50d9afed70b9ffcfd9183\nAuthor: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\nDate:   Thu Aug 18 01:00:41 2016 +0200\n\n    bpf: get rid of cgroup helper related ifdefs\n\n    As recently discussed during the task_under_cgroup_hierarchy() addition,\n    we should get rid of the ifdefs surrounding the bpf_skb_under_cgroup()\n    helper. If related functionality is not built-in, the helper cannot be\n    used anyway, which is also in line with what we do for all other helpers.\n\n    Signed-off-by: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Acked-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 49604d19727ab335184ce50c08cb4787175c0a77\nAuthor: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\nDate:   Thu Aug 18 01:00:40 2016 +0200\n\n    bpf: enable event output helper also for xdp types\n\n    Follow-up to 555c8a8623a3 (\"bpf: avoid stack copy and use skb ctx for\n    event output\") for also adding the event output helper for XDP typed\n    programs. The event output helper has been very useful in particular for\n    debugging or event notification purposes, since it\u0027s much faster and\n    flexible than regular trace printk due to programmatically being able to\n    attach meta data. Same flags structure applies as with tc BPF programs.\n\n    Signed-off-by: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Acked-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 1525487e67bb1f699506af61a734945ec7356245\nAuthor: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\nDate:   Thu Aug 18 01:00:39 2016 +0200\n\n    bpf: add bpf_skb_change_tail helper\n\n    This work adds a bpf_skb_change_tail() helper for tc BPF programs. The\n    basic idea is to expand or shrink the skb in a controlled manner. The\n    eBPF program can then rewrite the rest via helpers like bpf_skb_store_bytes(),\n    bpf_lX_csum_replace() and others rather than passing a raw buffer for\n    writing here.\n\n    bpf_skb_change_tail() is really a slow path helper and intended for\n    replies with f.e. ICMP control messages. Concept is similar to other\n    helpers like bpf_skb_change_proto() helper to keep the helper without\n    protocol specifics and let the BPF program mangle the remaining parts.\n    A flags field has been added and is reserved for now should we extend\n    the helper in future.\n\n    Signed-off-by: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Acked-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 3d0ac151eb1b8dd013bb7a149360959c0ae60f01\nAuthor: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\nDate:   Thu Aug 18 01:00:38 2016 +0200\n\n    bpf: use skb_pkt_type_ok helper in bpf_skb_change_type\n\n    Since we have a skb_pkt_type_ok() helper for checking the type before\n    mangling, make use of it instead of open coding. Follow-up to commit\n    8b10cab64c13 (\"net: simplify and make pkt_type_ok() available for other\n    users\") that came in after d2485c4242a8 (\"bpf: add bpf_skb_change_type\n    helper\").\n\n    Signed-off-by: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Acked-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit ab67f4e1fecde3adbc512baa7360cedbc44deac0\nAuthor: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\nDate:   Thu Aug 11 21:38:37 2016 +0200\n\n    bpf: fix write helpers with regards to non-linear parts\n\n    Fix the bpf_try_make_writable() helper and all call sites we have in BPF,\n    it\u0027s currently defect with regards to skbs when the write_len spans into\n    non-linear parts, no matter if cloned or not.\n\n    There are multiple issues at once. First, using skb_store_bits() is not\n    correct since even if we have a cloned skb, page frags can still be shared.\n    To really make them private, we need to pull them in via __pskb_pull_tail()\n    first, which also gets us a private head via pskb_expand_head() implicitly.\n\n    This is for helpers like bpf_skb_store_bytes(), bpf_l3_csum_replace(),\n    bpf_l4_csum_replace(). Really, the only thing reasonable and working here\n    is to call skb_ensure_writable() before any write operation. Meaning, via\n    pskb_may_pull() it makes sure that parts we want to access are pulled in and\n    if not does so plus unclones the skb implicitly. If our write_len still fits\n    the headlen and we\u0027re cloned and our header of the clone is not writable,\n    then we need to make a private copy via pskb_expand_head(). skb_store_bits()\n    is a bit misleading and only safe to store into non-linear data in different\n    contexts such as 357b40a18b04 (\"[IPV6]: IPV6_CHECKSUM socket option can\n    corrupt kernel memory\").\n\n    For above BPF helper functions, it means after fixed bpf_try_make_writable(),\n    we\u0027ve pulled in enough, so that we operate always based on skb-\u003edata. Thus,\n    the call to skb_header_pointer() and skb_store_bits() becomes superfluous.\n    In bpf_skb_store_bytes(), the len check is unnecessary too since it can\n    only pass in maximum of BPF stack size, so adding offset is guaranteed to\n    never overflow. Also bpf_l3/4_csum_replace() helpers must test for proper\n    offset alignment since they use __sum16 pointer for writing resulting csum.\n\n    The remaining helpers that change skb data not discussed here yet are\n    bpf_skb_vlan_push(), bpf_skb_vlan_pop() and bpf_skb_change_proto(). The\n    vlan helpers internally call either skb_ensure_writable() (pop case) and\n    skb_cow_head() (push case, for head expansion), respectively. Similarly,\n    bpf_skb_proto_xlat() takes care to not mangle page frags.\n\n    Fixes: 608cd71a9c7c (\"tc: bpf: generalize pedit action\")\n    Fixes: 91bc4822c3d6 (\"tc: bpf: add checksum helpers\")\n    Fixes: 3697649ff29e (\"bpf: try harder on clones when writing into skb\")\n    Signed-off-by: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Acked-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 1f476a566de499a6e11f27f46a116ee0fcf60b9d\nAuthor: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\nDate:   Tue Sep 20 00:26:14 2016 +0200\n\n    bpf: add test cases for direct packet access\n\n    Add couple of test cases for direct write and the negative size issue, and\n    also adjust the direct packet access test4 since it asserts that writes are\n    not possible, but since we\u0027ve just added support for writes, we need to\n    invert the verdict to ACCEPT, of course. Summary: 133 PASSED, 0 FAILED.\n\n    Signed-off-by: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Acked-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 8e932e250e06f391e708b92137c5a23d22938619\nAuthor: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\nDate:   Thu Sep 8 01:03:42 2016 +0200\n\n    bpf: fix range propagation on direct packet access\n\n    LLVM can generate code that tests for direct packet access via\n    skb-\u003edata/data_end in a way that currently gets rejected by the\n    verifier, example:\n\n      [...]\n       7: (61) r3 \u003d *(u32 *)(r6 +80)\n       8: (61) r9 \u003d *(u32 *)(r6 +76)\n       9: (bf) r2 \u003d r9\n      10: (07) r2 +\u003d 54\n      11: (3d) if r3 \u003e\u003d r2 goto pc+12\n       R1\u003dinv R2\u003dpkt(id\u003d0,off\u003d54,r\u003d0) R3\u003dpkt_end R4\u003dinv R6\u003dctx\n       R9\u003dpkt(id\u003d0,off\u003d0,r\u003d0) R10\u003dfp\n      12: (18) r4 \u003d 0xffffff7a\n      14: (05) goto pc+430\n      [...]\n\n      from 11 to 24: R1\u003dinv R2\u003dpkt(id\u003d0,off\u003d54,r\u003d0) R3\u003dpkt_end R4\u003dinv\n                     R6\u003dctx R9\u003dpkt(id\u003d0,off\u003d0,r\u003d0) R10\u003dfp\n      24: (7b) *(u64 *)(r10 -40) \u003d r1\n      25: (b7) r1 \u003d 0\n      26: (63) *(u32 *)(r6 +56) \u003d r1\n      27: (b7) r2 \u003d 40\n      28: (71) r8 \u003d *(u8 *)(r9 +20)\n      invalid access to packet, off\u003d20 size\u003d1, R9(id\u003d0,off\u003d0,r\u003d0)\n\n    The reason why this gets rejected despite a proper test is that we\n    currently call find_good_pkt_pointers() only in case where we detect\n    tests like rX \u003e pkt_end, where rX is of type pkt(id\u003dY,off\u003dZ,r\u003d0) and\n    derived, for example, from a register of type pkt(id\u003dY,off\u003d0,r\u003d0)\n    pointing to skb-\u003edata. find_good_pkt_pointers() then fills the range\n    in the current branch to pkt(id\u003dY,off\u003d0,r\u003dZ) on success.\n\n    For above case, we need to extend that to recognize pkt_end \u003e\u003d rX\n    pattern and mark the other branch that is taken on success with the\n    appropriate pkt(id\u003dY,off\u003d0,r\u003dZ) type via find_good_pkt_pointers().\n    Since eBPF operates on BPF_JGT (\u003e) and BPF_JGE (\u003e\u003d), these are the\n    only two practical options to test for from what LLVM could have\n    generated, since there\u0027s no such thing as BPF_JLT (\u003c) or BPF_JLE (\u003c\u003d)\n    that we would need to take into account as well.\n\n    After the fix:\n\n      [...]\n       7: (61) r3 \u003d *(u32 *)(r6 +80)\n       8: (61) r9 \u003d *(u32 *)(r6 +76)\n       9: (bf) r2 \u003d r9\n      10: (07) r2 +\u003d 54\n      11: (3d) if r3 \u003e\u003d r2 goto pc+12\n       R1\u003dinv R2\u003dpkt(id\u003d0,off\u003d54,r\u003d0) R3\u003dpkt_end R4\u003dinv R6\u003dctx\n       R9\u003dpkt(id\u003d0,off\u003d0,r\u003d0) R10\u003dfp\n      12: (18) r4 \u003d 0xffffff7a\n      14: (05) goto pc+430\n      [...]\n\n      from 11 to 24: R1\u003dinv R2\u003dpkt(id\u003d0,off\u003d54,r\u003d54) R3\u003dpkt_end R4\u003dinv\n                     R6\u003dctx R9\u003dpkt(id\u003d0,off\u003d0,r\u003d54) R10\u003dfp\n      24: (7b) *(u64 *)(r10 -40) \u003d r1\n      25: (b7) r1 \u003d 0\n      26: (63) *(u32 *)(r6 +56) \u003d r1\n      27: (b7) r2 \u003d 40\n      28: (71) r8 \u003d *(u8 *)(r9 +20)\n      29: (bf) r1 \u003d r8\n      30: (25) if r8 \u003e 0x3c goto pc+47\n       R1\u003dinv56 R2\u003dimm40 R3\u003dpkt_end R4\u003dinv R6\u003dctx R8\u003dinv56\n       R9\u003dpkt(id\u003d0,off\u003d0,r\u003d54) R10\u003dfp\n      31: (b7) r1 \u003d 1\n      [...]\n\n    Verifier test cases are also added in this work, one that demonstrates\n    the mentioned example here and one that tries a bad packet access for\n    the current/fall-through branch (the one with types pkt(id\u003dX,off\u003dY,r\u003d0),\n    pkt(id\u003dX,off\u003d0,r\u003d0)), then a case with good and bad accesses, and two\n    with both test variants (\u003e, \u003e\u003d).\n\n    Fixes: 969bf05eb3ce (\"bpf: direct packet access\")\n    Signed-off-by: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Acked-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 873b1be73c1d8a7d86a99c7005f6e7cec8568690\nAuthor: Aaron Yue \u003chaoxuany@fb.com\u003e\nDate:   Thu Aug 11 18:17:17 2016 -0700\n\n    samples/bpf: add verifier tests for the helper access to the packet\n\n    test various corner cases of the helper function access to the packet\n    via crafted XDP programs.\n\n    Signed-off-by: Aaron Yue \u003chaoxuany@fb.com\u003e\n    Signed-off-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Acked-by: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit dec7ab59647a8d197cc252b027f5533d2befb9a3\nAuthor: Alexei Starovoitov \u003cast@fb.com\u003e\nDate:   Thu May 5 19:49:15 2016 -0700\n\n    samples/bpf: add verifier tests\n\n    add few tests for \"pointer to packet\" logic of the verifier\n\n    Signed-off-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 5d13af94b66a9fdd4c98098c13baa79bba8de1ed\nAuthor: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\nDate:   Wed Apr 13 00:10:54 2016 +0200\n\n    bpf, samples: add test cases for raw stack\n\n    This adds test cases mostly around ARG_PTR_TO_RAW_STACK to check the\n    verifier behaviour.\n\n      [...]\n      #84 raw_stack: no skb_load_bytes OK\n      #85 raw_stack: skb_load_bytes, no init OK\n      #86 raw_stack: skb_load_bytes, init OK\n      #87 raw_stack: skb_load_bytes, spilled regs around bounds OK\n      #88 raw_stack: skb_load_bytes, spilled regs corruption OK\n      #89 raw_stack: skb_load_bytes, spilled regs corruption 2 OK\n      #90 raw_stack: skb_load_bytes, spilled regs + data OK\n      #91 raw_stack: skb_load_bytes, invalid access 1 OK\n      #92 raw_stack: skb_load_bytes, invalid access 2 OK\n      #93 raw_stack: skb_load_bytes, invalid access 3 OK\n      #94 raw_stack: skb_load_bytes, invalid access 4 OK\n      #95 raw_stack: skb_load_bytes, invalid access 5 OK\n      #96 raw_stack: skb_load_bytes, invalid access 6 OK\n      #97 raw_stack: skb_load_bytes, large access OK\n      Summary: 98 PASSED, 0 FAILED\n\n    Signed-off-by: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Acked-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 7e818ab049fd6f3f945baadb7ec9c9e71d0f6760\nAuthor: Alexei Starovoitov \u003cast@fb.com\u003e\nDate:   Mon Mar 7 21:57:20 2016 -0800\n\n    samples/bpf: add map_flags to bpf loader\n\n    note old loader is compatible with new kernel.\n    map_flags are optional\n\n    Signed-off-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit bb5568f8e3f8be87500558fc7409b5a2180e8196\nAuthor: tom.leiming@gmail.com \u003ctom.leiming@gmail.com\u003e\nDate:   Mon Feb 1 22:39:57 2016 -0800\n\n    samples/bpf: unit test for BPF_MAP_TYPE_PERCPU_ARRAY\n\n    A sanity test for BPF_MAP_TYPE_PERCPU_ARRAY\n\n    Signed-off-by: Ming Lei \u003ctom.leiming@gmail.com\u003e\n    Signed-off-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 8f013495498b22664c2e215696900b9a4787bd0b\nAuthor: Alexei Starovoitov \u003cast@fb.com\u003e\nDate:   Mon Mar 7 21:57:18 2016 -0800\n\n    samples/bpf: make map creation more verbose\n\n    map creation is typically the first one to fail when rlimits are\n    too low, not enough memory, etc\n    Make this failure scenario more verbose\n\n    Signed-off-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 71e38cc3478ffe6f15733a649c70db307e75fad6\nAuthor: Martin KaFai Lau \u003ckafai@fb.com\u003e\nDate:   Mon Feb 1 22:39:56 2016 -0800\n\n    samples/bpf: unit test for BPF_MAP_TYPE_PERCPU_HASH\n\n    A sanity test for BPF_MAP_TYPE_PERCPU_HASH.\n\n    Signed-off-by: Martin KaFai Lau \u003ckafai@fb.com\u003e\n    Signed-off-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit dd858c4846883fbde61bfa221f85ee04549254b5\nAuthor: Alexei Starovoitov \u003cast@fb.com\u003e\nDate:   Thu Sep 1 18:37:23 2016 -0700\n\n    bpf: perf_event progs should only use preallocated maps\n\n    Make sure that BPF_PROG_TYPE_PERF_EVENT programs only use\n    preallocated hash maps, since doing memory allocation\n    in overflow_handler can crash depending on where nmi got triggered.\n\n    Signed-off-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Acked-by: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 1b9b54d940e2aa7c51b97e8e5f26a1c332f83f96\nAuthor: Alexei Starovoitov \u003cast@fb.com\u003e\nDate:   Thu Sep 1 18:37:21 2016 -0700\n\n    bpf: support 8-byte metafield access\n\n    The verifier supported only 4-byte metafields in\n    struct __sk_buff and struct xdp_md. The metafields in upcoming\n    struct bpf_perf_event are 8-byte to match register width in struct pt_regs.\n    Teach verifier to recognize 8-byte metafield access.\n    The patch doesn\u0027t affect safety of sockets and xdp programs.\n    They check for 4-byte only ctx access before these conditions are hit.\n\n    Signed-off-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Acked-by: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit e317791173a2653deb2117cf6c6ec2525803c78b\nAuthor: Alexei Starovoitov \u003cast@fb.com\u003e\nDate:   Thu Aug 11 18:17:16 2016 -0700\n\n    bpf: allow helpers access the packet directly\n\n    The helper functions like bpf_map_lookup_elem(map, key) were only\n    allowing \u0027key\u0027 to point to the initialized stack area.\n    That is causing performance degradation when programs need to process\n    millions of packets per second and need to copy contents of the packet\n    into the stack just to pass the stack pointer into the lookup() function.\n    Allow such helpers read from the packet directly.\n    All helpers that expect ARG_PTR_TO_MAP_KEY, ARG_PTR_TO_MAP_VALUE,\n    ARG_PTR_TO_STACK assume byte aligned pointer, so no alignment concerns,\n    only need to check that helper will not be accessing beyond\n    the packet range verified by the prior \u0027if (ptr \u003c data_end)\u0027 condition.\n    For now allow this feature for XDP programs only. Later it can be\n    relaxed for the clsact programs as well.\n\n    Signed-off-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Acked-by: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit ee956448cd428ad0d5bb950753b4030dff4fa385\nAuthor: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\nDate:   Fri Aug 12 22:17:17 2016 +0200\n\n    bpf: fix bpf_skb_in_cgroup helper naming\n\n    While hashing out BPF\u0027s current_task_under_cgroup helper bits, it came\n    to discussion that the skb_in_cgroup helper name was suboptimally chosen.\n\n    Tejun says:\n\n      So, I think in_cgroup should mean that the object is in that\n      particular cgroup while under_cgroup in the subhierarchy of that\n      cgroup. Let\u0027s rename the other subhierarchy test to under too. I\n      think that\u0027d be a lot less confusing going forward.\n\n      [...]\n\n      It\u0027s more intuitive and gives us the room to implement the real\n      \"in\" test if ever necessary in the future.\n\n    Since this touches uapi bits, we need to change this as long as v4.8\n    is not yet officially released. Thus, change the helper enum and rename\n    related bits.\n\n    Fixes: 4a482f34afcc (\"cgroup: bpf: Add bpf_skb_in_cgroup_proto\")\n    Reference: http://patchwork.ozlabs.org/patch/658500/\n    Suggested-by: Sargun Dhillon \u003csargun@sargun.me\u003e\n    Suggested-by: Tejun Heo \u003ctj@kernel.org\u003e\n    Signed-off-by: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Acked-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 9a6848be325ec1f65ecda69f29a0f7de6ef11583\nAuthor: Martin KaFai Lau \u003ckafai@fb.com\u003e\nDate:   Thu Jun 30 10:28:45 2016 -0700\n\n    cgroup: bpf: Add an example to do cgroup checking in BPF\n\n    test_cgrp2_array_pin.c:\n    A userland program that creates a bpf_map (BPF_MAP_TYPE_GROUP_ARRAY),\n    pouplates/updates it with a cgroup2\u0027s backed fd and pins it to a\n    bpf-fs\u0027s file.  The pinned file can be loaded by tc and then used\n    by the bpf prog later.  This program can also update an existing pinned\n    array and it could be useful for debugging/testing purpose.\n\n    test_cgrp2_tc_kern.c:\n    A bpf prog which should be loaded by tc.  It is to demonstrate\n    the usage of bpf_skb_in_cgroup.\n\n    test_cgrp2_tc.sh:\n    A script that glues the test_cgrp2_array_pin.c and\n    test_cgrp2_tc_kern.c together.  The idea is like:\n    1. Load the test_cgrp2_tc_kern.o by tc\n    2. Use test_cgrp2_array_pin.c to populate a BPF_MAP_TYPE_CGROUP_ARRAY\n       with a cgroup fd\n    3. Do a \u0027ping -6 ff02::1%ve\u0027 to ensure the packet has been\n       dropped because of a match on the cgroup\n\n    Most of the lines in test_cgrp2_tc.sh is the boilerplate\n    to setup the cgroup/bpf-fs/net-devices/netns...etc.  It is\n    not bulletproof on errors but should work well enough and\n    give enough debug info if things did not go well.\n\n    Signed-off-by: Martin KaFai Lau \u003ckafai@fb.com\u003e\n    Cc: Alexei Starovoitov \u003cast@fb.com\u003e\n    Cc: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Cc: Tejun Heo \u003ctj@kernel.org\u003e\n    Acked-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 554c0c2a1e638f804abd90ec1a6465b944ebc43a\nAuthor: Alexei Starovoitov \u003cast@fb.com\u003e\nDate:   Thu May 5 19:49:14 2016 -0700\n\n    samples/bpf: add \u0027pointer to packet\u0027 tests\n\n    parse_simple.c - packet parser exapmle with single length check that\n    filters out udp packets for port 9\n\n    parse_varlen.c - variable length parser that understand multiple vlan headers,\n    ipip, ipip6 and ip options to filter out udp or tcp packets on port 9.\n    The packet is parsed layer by layer with multitple length checks.\n\n    parse_ldabs.c - classic style of packet parsing using LD_ABS instruction.\n    Same functionality as parse_simple.\n\n    simple \u003d 24.1Mpps per core\n    varlen \u003d 22.7Mpps\n    ldabs  \u003d 21.4Mpps\n\n    Parser with LD_ABS instructions is slower than full direct access parser\n    which does more packet accesses and checks.\n\n    These examples demonstrate the choice bpf program authors can make between\n    flexibility of the parser vs speed.\n\n    Signed-off-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 05c11618658607b7b87630b177ae360cd3832108\nAuthor: Alexei Starovoitov \u003cast@fb.com\u003e\nDate:   Wed Apr 6 18:43:31 2016 -0700\n\n    samples/bpf: add tracepoint vs kprobe performance tests\n\n    the first microbenchmark does\n    fd\u003dopen(\"/proc/self/comm\");\n    for() {\n      write(fd, \"test\");\n    }\n    and on 4 cpus in parallel:\n                                          writes per sec\n    base (no tracepoints, no kprobes)         930k\n    with kprobe at __set_task_comm()          420k\n    with tracepoint at task:task_rename       730k\n\n    For kprobe + full bpf program manully fetches oldcomm, newcomm via bpf_probe_read.\n    For tracepint bpf program does nothing, since arguments are copied by tracepoint.\n\n    2nd microbenchmark does:\n    fd\u003dopen(\"/dev/urandom\");\n    for() {\n      read(fd, buf);\n    }\n    and on 4 cpus in parallel:\n                                           reads per sec\n    base (no tracepoints, no kprobes)         300k\n    with kprobe at urandom_read()             279k\n    with tracepoint at random:urandom_read    290k\n\n    bpf progs attached to kprobe and tracepoint are noop.\n\n    Signed-off-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 11304e9b3f6f30bde8203e8a40bc07d1484026a1\nAuthor: Alexei Starovoitov \u003cast@fb.com\u003e\nDate:   Tue Mar 8 15:07:54 2016 -0800\n\n    samples/bpf: add map performance test\n\n    performance tests for hash map and per-cpu hash map\n    with and without pre-allocation\n\n    Signed-off-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 619046c6d05171bcb9289345c661a127e5838eae\nAuthor: Alexei Starovoitov \u003cast@fb.com\u003e\nDate:   Tue Mar 8 15:07:52 2016 -0800\n\n    samples/bpf: add bpf map stress test\n\n    this test calls bpf programs from different contexts:\n    from inside of slub, from rcu, from pretty much everywhere,\n    since it kprobes all spin_lock functions.\n    It stresses the bpf hash and percpu map pre-allocation,\n    deallocation logic and call_rcu mechanisms.\n    User space part adding more stress by walking and deleting map elements.\n\n    Note that due to nature bpf_load.c the earlier kprobe+bpf programs are\n    already active while loader loads new programs, creates new kprobes and\n    attaches them.\n\n    Signed-off-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit edb511a8eed3a35d263e3b41b656d0d2e042b2f1\nAuthor: Sargun Dhillon \u003csargun@sargun.me\u003e\nDate:   Fri Aug 12 08:56:52 2016 -0700\n\n    bpf: Add bpf_current_task_under_cgroup helper\n\n    This adds a bpf helper that\u0027s similar to the skb_in_cgroup helper to check\n    whether the probe is currently executing in the context of a specific\n    subset of the cgroupsv2 hierarchy. It does this based on membership test\n    for a cgroup arraymap. It is invalid to call this in an interrupt, and\n    it\u0027ll return an error. The helper is primarily to be used in debugging\n    activities for containers, where you may have multiple programs running in\n    a given top-level \"container\".\n\n    Signed-off-by: Sargun Dhillon \u003csargun@sargun.me\u003e\n    Cc: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Cc: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Cc: Tejun Heo \u003ctj@kernel.org\u003e\n    Acked-by: Tejun Heo \u003ctj@kernel.org\u003e\n    Acked-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Acked-by: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 6d795673a81321dddaabd489fc5cb797be9bb179\nAuthor: Sargun Dhillon \u003csargun@sargun.me\u003e\nDate:   Mon Jul 25 05:54:46 2016 -0700\n\n    bpf: Add bpf_probe_write_user BPF helper to be called in tracers\n\n    This allows user memory to be written to during the course of a kprobe.\n    It shouldn\u0027t be used to implement any kind of security mechanism\n    because of TOC-TOU attacks, but rather to debug, divert, and\n    manipulate execution of semi-cooperative processes.\n\n    Although it uses probe_kernel_write, we limit the address space\n    the probe can write into by checking the space with access_ok.\n    We do this as opposed to calling copy_to_user directly, in order\n    to avoid sleeping. In addition we ensure the threads\u0027s current fs\n    / segment is USER_DS and the thread isn\u0027t exiting nor a kernel thread.\n\n    Given this feature is meant for experiments, and it has a risk of\n    crashing the system, and running programs, we print a warning on\n    when a proglet that attempts to use this helper is installed,\n    along with the pid and process name.\n\n    Signed-off-by: Sargun Dhillon \u003csargun@sargun.me\u003e\n    Cc: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Cc: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Acked-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit c55d6a90647e422426377f262f39fb6d5fe756b2\nAuthor: Alexei Starovoitov \u003cast@fb.com\u003e\nDate:   Wed Feb 17 19:58:59 2016 -0800\n\n    samples/bpf: offwaketime example\n\n    This is simplified version of Brendan Gregg\u0027s offwaketime:\n    This program shows kernel stack traces and task names that were blocked and\n    \"off-CPU\", along with the stack traces and task names for the threads that woke\n    them, and the total elapsed time from when they blocked to when they were woken\n    up. The combined stacks, task names, and total time is summarized in kernel\n    context for efficiency.\n\n    Example:\n    $ sudo ./offwaketime | flamegraph.pl \u003e demo.svg\n    Open demo.svg in the browser as FlameGraph visualization.\n\n    Signed-off-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit b7d3d530ad8d254eddd35367b1deede6c67dc2fc\nAuthor: Andrew Morton \u003cakpm@linux-foundation.org\u003e\nDate:   Mon Jul 18 15:50:58 2016 -0700\n\n    kernel/trace/bpf_trace.c: work around gcc-4.4.4 anon union initialization bug\n\n    kernel/trace/bpf_trace.c: In function \u0027bpf_event_output\u0027:\n    kernel/trace/bpf_trace.c:312: error: unknown field \u0027next\u0027 specified in initializer\n    kernel/trace/bpf_trace.c:312: warning: missing braces around initializer\n    kernel/trace/bpf_trace.c:312: warning: (near initialization for \u0027raw.frag.\u003canonymous\u003e\u0027)\n\n    Fixes: 555c8a8623a3a87 (\"bpf: avoid stack copy and use skb ctx for event output\")\n    Acked-by: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Cc: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Cc: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Andrew Morton \u003cakpm@linux-foundation.org\u003e\n    Acked-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit c26cfd27cfcd02b861e76cd3b161ffca56a50c4f\nAuthor: Alexei Starovoitov \u003cast@fb.com\u003e\nDate:   Wed Jul 6 22:38:36 2016 -0700\n\n    bpf: introduce bpf_get_current_task() helper\n\n    over time there were multiple requests to access different data\n    structures and fields of task_struct current, so finally add\n    the helper to access \u0027current\u0027 as-is. Tracing bpf programs will do\n    the rest of walking the pointers via bpf_probe_read().\n    Note that current can be null and bpf program has to deal it with,\n    but even dumb passing null into bpf_probe_read() is still safe.\n\n    Suggested-by: Brendan Gregg \u003cbrendan.d.gregg@gmail.com\u003e\n    Signed-off-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Acked-by: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit fbea0d499e0041c6ecffc1e0561b7f7648f63873\nAuthor: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\nDate:   Sun Jul 3 01:28:47 2016 +0200\n\n    bpf: add bpf_get_hash_recalc helper\n\n    If skb_clear_hash() was invoked due to mangling of relevant headers and\n    BPF program needs skb-\u003ehash later on, we can add a helper to trigger hash\n    recalculation via bpf_get_hash_recalc().\n\n    The helper will return the newly retrieved hash directly, but later access\n    can also be done via skb context again through skb-\u003ehash directly (inline)\n    without needing to call the helper once more.\n\n    Signed-off-by: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Acked-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit d46c43b45b9139adbffdeacb03fd14eca7cc5220\nAuthor: Alexei Starovoitov \u003cast@fb.com\u003e\nDate:   Fri Aug 5 14:01:27 2016 -0700\n\n    bpf: restore behavior of bpf_map_update_elem\n\n    The introduction of pre-allocated hash elements inadvertently broke\n    the behavior of bpf hash maps where users expected to call\n    bpf_map_update_elem() without considering that the map can be full.\n    Some programs do:\n    old_value \u003d bpf_map_lookup_elem(map, key);\n    if (old_value) {\n      ... prepare new_value on stack ...\n      bpf_map_update_elem(map, key, new_value);\n    }\n    Before pre-alloc the update() for existing element would work even\n    in \u0027map full\u0027 condition. Restore this behavior.\n\n    The above program could have updated old_value in place instead of\n    update() which would be faster and most programs use that approach,\n    but sometimes the values are large and the programs use update()\n    helper to do atomic replacement of the element.\n    Note we cannot simply update element\u0027s value in-place like percpu\n    hash map does and have to allocate extra num_possible_cpu elements\n    and use this extra reserve when the map is full.\n\n    Fixes: 6c9059817432 (\"bpf: pre-allocate hash map elements\")\n    Signed-off-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 4527c21762060ff534949c16348edb1cb0c53ce3\nAuthor: Jakub Kicinski \u003cjakub.kicinski@netronome.com\u003e\nDate:   Tue Aug 2 16:12:14 2016 +0100\n\n    bpf: fix method of PTR_TO_PACKET reg id generation\n\n    Using per-register incrementing ID can lead to\n    find_good_pkt_pointers() confusing registers which\n    have completely different values.  Consider example:\n\n    0: (bf) r6 \u003d r1\n    1: (61) r8 \u003d *(u32 *)(r6 +76)\n    2: (61) r0 \u003d *(u32 *)(r6 +80)\n    3: (bf) r7 \u003d r8\n    4: (07) r8 +\u003d 32\n    5: (2d) if r8 \u003e r0 goto pc+9\n     R0\u003dpkt_end R1\u003dctx R6\u003dctx R7\u003dpkt(id\u003d0,off\u003d0,r\u003d32) R8\u003dpkt(id\u003d0,off\u003d32,r\u003d32) R10\u003dfp\n    6: (bf) r8 \u003d r7\n    7: (bf) r9 \u003d r7\n    8: (71) r1 \u003d *(u8 *)(r7 +0)\n    9: (0f) r8 +\u003d r1\n    10: (71) r1 \u003d *(u8 *)(r7 +1)\n    11: (0f) r9 +\u003d r1\n    12: (07) r8 +\u003d 32\n    13: (2d) if r8 \u003e r0 goto pc+1\n     R0\u003dpkt_end R1\u003dinv56 R6\u003dctx R7\u003dpkt(id\u003d0,off\u003d0,r\u003d32) R8\u003dpkt(id\u003d1,off\u003d32,r\u003d32) R9\u003dpkt(id\u003d1,off\u003d0,r\u003d32) R10\u003dfp\n    14: (71) r1 \u003d *(u8 *)(r9 +16)\n    15: (b7) r7 \u003d 0\n    16: (bf) r0 \u003d r7\n    17: (95) exit\n\n    We need to get a UNKNOWN_VALUE with imm to force id\n    generation so lines 0-5 make r7 a valid packet pointer.\n    We then read two different bytes from the packet and\n    add them to copies of the constructed packet pointer.\n    r8 (line 9) and r9 (line 11) will get the same id of 1,\n    independently.  When either of them is validated (line\n    13) - find_good_pkt_pointers() will also mark the other\n    as safe.  This leads to access on line 14 being mistakenly\n    considered safe.\n\n    Fixes: 969bf05eb3ce (\"bpf: direct packet access\")\n    Signed-off-by: Jakub Kicinski \u003cjakub.kicinski@netronome.com\u003e\n    Acked-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Acked-by: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit bb13276696f4411d2401df5cf25f26bbe82d96ee\nAuthor: Brenden Blanco \u003cbblanco@plumgrid.com\u003e\nDate:   Tue Jul 19 12:16:56 2016 -0700\n\n    bpf: enable direct packet data write for xdp progs\n\n    For forwarding to be effective, XDP programs should be allowed to\n    rewrite packet data.\n\n    This requires that the drivers supporting XDP must all map the packet\n    memory as TODEVICE or BIDIRECTIONAL before invoking the program.\n\n    Signed-off-by: Brenden Blanco \u003cbblanco@plumgrid.com\u003e\n    Acked-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 59611c698e67d38393f4c71322a5465d3c4a6920\nAuthor: Brenden Blanco \u003cbblanco@plumgrid.com\u003e\nDate:   Tue Jul 19 12:16:47 2016 -0700\n\n    bpf: add XDP prog type for early driver filter\n\n    Add a new bpf prog type that is intended to run in early stages of the\n    packet rx path. Only minimal packet metadata will be available, hence a\n    new context type, struct xdp_md, is exposed to userspace. So far only\n    expose the packet start and end pointers, and only in read mode.\n\n    An XDP program must return one of the well known enum values, all other\n    return codes are reserved for future use. Unfortunately, this\n    restriction is hard to enforce at verification time, so take the\n    approach of warning at runtime when such programs are encountered. Out\n    of bounds return codes should alias to XDP_ABORTED.\n\n    Signed-off-by: Brenden Blanco \u003cbblanco@plumgrid.com\u003e\n    Acked-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 291e76e5d2af73bfc0d3ce57e72d0e6ad691266c\nAuthor: Alexei Starovoitov \u003cast@fb.com\u003e\nDate:   Thu May 5 19:49:12 2016 -0700\n\n    bpf: wire in data and data_end for cls_act_bpf\n\n    allow cls_bpf and act_bpf programs access skb-\u003edata and skb-\u003edata_end pointers.\n    The bpf helpers that change skb-\u003edata need to update data_end pointer as well.\n    The verifier checks that programs always reload data, data_end pointers\n    after calls to such bpf helpers.\n    We cannot add \u0027data_end\u0027 pointer to struct qdisc_skb_cb directly,\n    since it\u0027s embedded as-is by infiniband ipoib, so wrapper struct is needed.\n\n    Signed-off-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Acked-by: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit f28b5534232c21bbac6b8abc922c5c6fd355a892\nAuthor: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\nDate:   Wed Jan 6 22:32:16 2016 +0100\n\n    bpf: cleanup bpf_prog_run_{save,clear}_cb helpers\n\n    Move the details behind the cb[] access into a small helper to decouple\n    and make them generic for bpf_prog_run_save_cb()/bpf_prog_run_clear_cb()\n    that was introduced via commit ff936a04e5f2 (\"bpf: fix cb access in socket\n    filter programs\"). Also add a comment to better clarify what is done in\n    bpf_skb_cb().\n\n    Signed-off-by: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Acked-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit bb9f86321da94643118a9183610874070b66659b\nAuthor: Brenden Blanco \u003cbblanco@plumgrid.com\u003e\nDate:   Tue Jul 19 12:16:46 2016 -0700\n\n    bpf: add bpf_prog_add api for bulk prog refcnt\n\n    A subsystem may need to store many copies of a bpf program, each\n    deserving its own reference. Rather than requiring the caller to loop\n    one by one (with possible mid-loop failure), add a bulk bpf_prog_add\n    api.\n\n    Signed-off-by: Brenden Blanco \u003cbblanco@plumgrid.com\u003e\n    Acked-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit bd0470638e184265fa12286af3d2794b8adfa033\nAuthor: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\nDate:   Sat Jul 16 01:15:55 2016 +0200\n\n    bpf: bpf_event_entry_gen\u0027s alloc needs to be in atomic context\n\n    Should have been obvious, only called from bpf() syscall via map_update_elem()\n    that calls bpf_fd_array_map_update_elem() under RCU read lock and thus this\n    must also be in GFP_ATOMIC, of course.\n\n    Fixes: 3b1efb196eee (\"bpf, maps: flush own entries on perf map release\")\n    Signed-off-by: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Acked-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit dd0994cece4f1885054b6394a890325c41ab6601\nAuthor: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\nDate:   Thu Jul 14 18:08:05 2016 +0200\n\n    bpf: avoid stack copy and use skb ctx for event output\n\n    This work addresses a couple of issues bpf_skb_event_output()\n    helper currently has: i) We need two copies instead of just a\n    single one for the skb data when it should be part of a sample.\n    The data can be non-linear and thus needs to be extracted via\n    bpf_skb_load_bytes() helper first, and then copied once again\n    into the ring buffer slot. ii) Since bpf_skb_load_bytes()\n    currently needs to be used first, the helper needs to see a\n    constant size on the passed stack buffer to make sure BPF\n    verifier can do sanity checks on it during verification time.\n    Thus, just passing skb-\u003elen (or any other non-constant value)\n    wouldn\u0027t work, but changing bpf_skb_load_bytes() is also not\n    the proper solution, since the two copies are generally still\n    needed. iii) bpf_skb_load_bytes() is just for rather small\n    buffers like headers, since they need to sit on the limited\n    BPF stack anyway. Instead of working around in bpf_skb_load_bytes(),\n    this work improves the bpf_skb_event_output() helper to address\n    all 3 at once.\n\n    We can make use of the passed in skb context that we have in\n    the helper anyway, and use some of the reserved flag bits as\n    a length argument. The helper will use the new __output_custom()\n    facility from perf side with bpf_skb_copy() as callback helper\n    to walk and extract the data. It will pass the data for setup\n    to bpf_event_output(), which generates and pushes the raw record\n    with an additional frag part. The linear data used in the first\n    frag of the record serves as programmatically defined meta data\n    passed along with the appended sample.\n\n    Signed-off-by: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Acked-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 1a4b7f9572753320d7f689c3b5452238b38444f0\nAuthor: Paul Gortmaker \u003cpaul.gortmaker@windriver.com\u003e\nDate:   Mon Jul 11 12:51:01 2016 -0400\n\n    bpf: make inode code explicitly non-modular\n\n    The Kconfig currently controlling compilation of this code is:\n\n    init/Kconfig:config BPF_SYSCALL\n    init/Kconfig:   bool \"Enable bpf() system call\"\n\n    ...meaning that it currently is not being built as a module by anyone.\n\n    Lets remove the couple traces of modular infrastructure use, so that\n    when reading the driver there is no doubt it is builtin-only.\n\n    Note that MODULE_ALIAS is a no-op for non-modular code.\n\n    We replace module.h with init.h since the file does use __init.\n\n    Cc: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Cc: netdev@vger.kernel.org\n    Signed-off-by: Paul Gortmaker \u003cpaul.gortmaker@windriver.com\u003e\n    Acked-by: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 50029f37fa71f61c3e3477ea6861414981d212eb\nAuthor: Martin KaFai Lau \u003ckafai@fb.com\u003e\nDate:   Thu Jun 30 10:28:44 2016 -0700\n\n    cgroup: bpf: Add bpf_skb_in_cgroup_proto\n\n    Adds a bpf helper, bpf_skb_in_cgroup, to decide if a skb-\u003esk\n    belongs to a descendant of a cgroup2.  It is similar to the\n    feature added in netfilter:\n    commit c38c4597e4bf (\"netfilter: implement xt_cgroup cgroup2 path match\")\n\n    The user is expected to populate a BPF_MAP_TYPE_CGROUP_ARRAY\n    which will be used by the bpf_skb_in_cgroup.\n\n    Modifications to the bpf verifier is to ensure BPF_MAP_TYPE_CGROUP_ARRAY\n    and bpf_skb_in_cgroup() are always used together.\n\n    Signed-off-by: Martin KaFai Lau \u003ckafai@fb.com\u003e\n    Cc: Alexei Starovoitov \u003cast@fb.com\u003e\n    Cc: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Cc: Tejun Heo \u003ctj@kernel.org\u003e\n    Acked-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit a307aca425ed5b6886185d386a85d6cc4a5fb268\nAuthor: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\nDate:   Tue Jun 28 12:18:28 2016 +0200\n\n    bpf: add bpf_skb_change_type helper\n\n    This work adds a helper for changing skb-\u003epkt_type in a controlled way.\n    We only allow a subset of possible values and can extend that in future\n    should other use cases come up. Doing this as a helper has the advantage\n    that errors can be handeled gracefully and thus helper kept extensible.\n\n    It\u0027s a write counterpart to pkt_type member we can already read from\n    struct __sk_buff context. Major use case is to change incoming skbs to\n    PACKET_HOST in a programmatic way instead of having to recirculate via\n    redirect(..., BPF_F_INGRESS), for example.\n\n    Signed-off-by: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Acked-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 33c60bb05b95b72b164b4035a0bb0dfa56e9c8e0\nAuthor: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\nDate:   Tue Jun 28 12:18:27 2016 +0200\n\n    bpf: add bpf_skb_change_proto helper\n\n    This patch adds a minimal helper for doing the groundwork of changing\n    the skb-\u003eprotocol in a controlled way. Currently supported is v4 to\n    v6 and vice versa transitions, which allows f.e. for a minimal, static\n    nat64 implementation where applications in containers that still\n    require IPv4 can be transparently operated in an IPv6-only environment.\n    For example, host facing veth of the container can transparently do\n    the transitions in a programmatic way with the help of clsact qdisc\n    and cls_bpf.\n\n    Idea is to separate concerns for keeping complexity of the helper\n    lower, which means that the programs utilize bpf_skb_change_proto(),\n    bpf_skb_store_bytes() and bpf_lX_csum_replace() to get the job done,\n    instead of doing everything in a single helper (and thus partially\n    duplicating helper functionality). Also, bpf_skb_change_proto()\n    shouldn\u0027t need to deal with raw packet data as this is done by other\n    helpers.\n\n    bpf_skb_proto_6_to_4() and bpf_skb_proto_4_to_6() unclone the skb to\n    operate on a private one, push or pop additionally required header\n    space and migrate the gso/gro meta data from the shared info. We do\n    mark the gso type as dodgy so that headers are checked and segs\n    recalculated by the gso/gro engine. The gso_size target is adapted\n    as well. The flags argument added is currently reserved and can be\n    used for future extensions.\n\n    Signed-off-by: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Acked-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 00df405522dd023da4649327bf4d212bc2137e8e\nAuthor: Martin KaFai Lau \u003ckafai@fb.com\u003e\nDate:   Thu Jun 30 10:28:43 2016 -0700\n\n    cgroup: bpf: Add BPF_MAP_TYPE_CGROUP_ARRAY\n\n    Add a BPF_MAP_TYPE_CGROUP_ARRAY and its bpf_map_ops\u0027s implementations.\n    To update an element, the caller is expected to obtain a cgroup2 backed\n    fd by open(cgroup2_dir) and then update the array with that fd.\n\n    Signed-off-by: Martin KaFai Lau \u003ckafai@fb.com\u003e\n    Cc: Alexei Starovoitov \u003cast@fb.com\u003e\n    Cc: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Cc: Tejun Heo \u003ctj@kernel.org\u003e\n    Acked-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 9717a3efc1196ce2bb34aa6bb6c90fd86df8b14b\nAuthor: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\nDate:   Thu Jun 30 17:24:44 2016 +0200\n\n    bpf: refactor bpf_prog_get and type check into helper\n\n    Since bpf_prog_get() and program type check is used in a couple of places,\n    refactor this into a small helper function that we can make use of. Since\n    the non RO prog-\u003eaux part is not used in performance critical paths and a\n    program destruction via RCU is rather very unlikley when doing the put, we\n    shouldn\u0027t have an issue just doing the bpf_prog_get() + prog-\u003etype !\u003d type\n    check, but actually not taking the ref at all (due to being in fdget() /\n    fdput() section of the bpf fd) is even cleaner and makes the diff smaller\n    as well, so just go for that. Callsites are changed to make use of the new\n    helper where possible.\n\n    Signed-off-by: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Acked-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 9e261a507bb8124fbaf141ab66c7da20ddf52d65\nAuthor: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\nDate:   Thu Jun 30 17:24:43 2016 +0200\n\n    bpf: generally move prog destruction to RCU deferral\n\n    Jann Horn reported following analysis that could potentially result\n    in a very hard to trigger (if not impossible) UAF race, to quote his\n    event timeline:\n\n     - Set up a process with threads T1, T2 and T3\n     - Let T1 set up a socket filter F1 that invokes another filter F2\n       through a BPF map [tail call]\n     - Let T1 trigger the socket filter via a unix domain socket write,\n       don\u0027t wait for completion\n     - Let T2 call PERF_EVENT_IOC_SET_BPF with F2, don\u0027t wait for completion\n     - Now T2 should be behind bpf_prog_get(), but before bpf_prog_put()\n     - Let T3 close the file descriptor for F2, dropping the reference\n       count of F2 to 2\n     - At this point, T1 should have looked up F2 from the map, but not\n       finished executing it\n     - Let T3 remove F2 from the BPF map, dropping the reference count of\n       F2 to 1\n     - Now T2 should call bpf_prog_put() (wrong BPF program type), dropping\n       the reference count of F2 to 0 and scheduling bpf_prog_free_deferred()\n       via schedule_work()\n     - At this point, the BPF program could be freed\n     - BPF execution is still running in a freed BPF program\n\n    While at PERF_EVENT_IOC_SET_BPF time it\u0027s only guaranteed that the perf\n    event fd we\u0027re doing the syscall on doesn\u0027t disappear from underneath us\n    for whole syscall time, it may not be the case for the bpf fd used as\n    an argument only after we did the put. It needs to be a valid fd pointing\n    to a BPF program at the time of the call to make the bpf_prog_get() and\n    while T2 gets preempted, F2 must have dropped reference to 1 on the other\n    CPU. The fput() from the close() in T3 should also add additionally delay\n    to the reference drop via exit_task_work() when bpf_prog_release() gets\n    called as well as scheduling bpf_prog_free_deferred().\n\n    That said, it makes nevertheless sense to move the BPF prog destruction\n    generally after RCU grace period to guarantee that such scenario above,\n    but also others as recently fixed in ceb56070359b (\"bpf, perf: delay release\n    of BPF prog after grace period\") with regards to tail calls won\u0027t happen.\n    Integrating bpf_prog_free_deferred() directly into the RCU callback is\n    not allowed since the invocation might happen from either softirq or\n    process context, so we\u0027re not permitted to block. Reviewing all bpf_prog_put()\n    invocations from eBPF side (note, cBPF -\u003e eBPF progs don\u0027t use this for\n    their destruction) with call_rcu() look good to me.\n\n    Since we don\u0027t know whether at the time of attaching the program, we\u0027re\n    already part of a tail call map, we need to use RCU variant. However, due\n    to this, there won\u0027t be severely more stress on the RCU callback queue:\n    situations with above bpf_prog_get() and bpf_prog_put() combo in practice\n    normally won\u0027t lead to releases, but even if they would, enough effort/\n    cycles have to be put into loading a BPF program into the kernel already.\n\n    Reported-by: Jann Horn \u003cjannh@google.com\u003e\n    Signed-off-by: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Acked-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 18c248c0ff7de1de9df39832a3fbc654d32a7b19\nAuthor: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\nDate:   Tue Jun 28 12:18:26 2016 +0200\n\n    bpf: don\u0027t use raw processor id in generic helper\n\n    Use smp_processor_id() for the generic helper bpf_get_smp_processor_id()\n    instead of the raw variant. This allows for preemption checks when we\n    have DEBUG_PREEMPT, and otherwise uses the raw variant anyway. We only\n    need to keep the raw variant for socket filters, but we can reuse the\n    helper that is already there from cBPF side.\n\n    Signed-off-by: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Acked-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 29b295cc2e5a945f8665a318ddf5c37e9d2cc6eb\nAuthor: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\nDate:   Tue Jun 28 12:18:23 2016 +0200\n\n    bpf: minor cleanups on fd maps and helpers\n\n    Some minor cleanups: i) Remove the unlikely() from fd array map lookups\n    and let the CPU branch predictor do its job, scenarios where there is not\n    always a map entry are very well valid. ii) Move the attribute type check\n    in the bpf_perf_event_read() helper a bit earlier so it\u0027s consistent wrt\n    checks with bpf_perf_event_output() helper as well. iii) remove some\n    comments that are self-documenting in kprobe_prog_is_valid_access() and\n    therefore make it consistent to tp_prog_is_valid_access() as well.\n\n    Signed-off-by: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Acked-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit bf05a1abe8d6579c08f2afe0c9e8ee0b61fa4256\nAuthor: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\nDate:   Wed Jun 15 22:47:14 2016 +0200\n\n    bpf, maps: flush own entries on perf map release\n\n    The behavior of perf event arrays are quite different from all\n    others as they are tightly coupled to perf event fds, f.e. shown\n    recently by commit e03e7ee34fdd (\"perf/bpf: Convert perf_event_array\n    to use struct file\") to make refcounting on perf event more robust.\n    A remaining issue that the current code still has is that since\n    additions to the perf event array take a reference on the struct\n    file via perf_event_get() and are only released via fput() (that\n    cleans up the perf event eventually via perf_event_release_kernel())\n    when the element is either manually removed from the map from user\n    space or automatically when the last reference on the perf event\n    map is dropped. However, this leads us to dangling struct file\u0027s\n    when the map gets pinned after the application owning the perf\n    event descriptor exits, and since the struct file reference will\n    in such case only be manually dropped or via pinned file removal,\n    it leads to the perf event living longer than necessary, consuming\n    needlessly resources for that time.\n\n    Relations between perf event fds and bpf perf event map fds can be\n    rather complex. F.e. maps can act as demuxers among different perf\n    event fds that can possibly be owned by different threads and based\n    on the index selection from the program, events get dispatched to\n    one of the per-cpu fd endpoints. One perf event fd (or, rather a\n    per-cpu set of them) can also live in multiple perf event maps at\n    the same time, listening for events. Also, another requirement is\n    that perf event fds can get closed from application side after they\n    have been attached to the perf event map, so that on exit perf event\n    map will take care of dropping their references eventually. Likewise,\n    when such maps are pinned, the intended behavior is that a user\n    application does bpf_obj_get(), puts its fds in there and on exit\n    when fd is released, they are dropped from the map again, so the map\n    acts rather as connector endpoint. This also makes perf event maps\n    inherently different from program arrays as described in more detail\n    in commit c9da161c6517 (\"bpf: fix clearing on persistent program\n    array maps\").\n\n    To tackle this, map entries are marked by the map struct file that\n    added the element to the map. And when the last reference to that map\n    struct file is released from user space, then the tracked entries\n    are purged from the map. This is okay, because new map struct files\n    instances resp. frontends to the anon inode are provided via\n    bpf_map_new_fd() that is called when we invoke bpf_obj_get_user()\n    for retrieving a pinned map, but also when an initial instance is\n    created via map_create(). The rest is resolved by the vfs layer\n    automatically for us by keeping reference count on the map\u0027s struct\n    file. Any concurrent updates on the map slot are fine as well, it\n    just means that perf_event_fd_array_release() needs to delete less\n    of its own entires.\n\n    Signed-off-by: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Acked-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 67fad7b94bbc2b0b65a4cb90072fc36802d29774\nAuthor: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\nDate:   Wed Jun 15 22:47:13 2016 +0200\n\n    bpf, maps: extend map_fd_get_ptr arguments\n\n    This patch extends map_fd_get_ptr() callback that is used by fd array\n    maps, so that struct file pointer from the related map can be passed\n    in. It\u0027s safe to remove map_update_elem() callback for the two maps since\n    this is only allowed from syscall side, but not from eBPF programs for these\n    two map types. Like in per-cpu map case, bpf_fd_array_map_update_elem()\n    needs to be called directly here due to the extra argument.\n\n    Signed-off-by: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Acked-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 163d05f1b8aac8e276ba238345ea55ed3d02db4a\nAuthor: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\nDate:   Wed Jun 15 22:47:12 2016 +0200\n\n    bpf, maps: add release callback\n\n    Add a release callback for maps that is invoked when the last\n    reference to its struct file is gone and the struct file about\n    to be released by vfs. The handler will be used by fd array maps.\n\n    Signed-off-by: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Acked-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 861cbc2d0d3987683f8276388a51ec750e5aa505\nAuthor: Alexei Starovoitov \u003cast@fb.com\u003e\nDate:   Wed Jun 15 18:25:38 2016 -0700\n\n    bpf: fix matching of data/data_end in verifier\n\n    The ctx structure passed into bpf programs is different depending on bpf\n    program type. The verifier incorrectly marked ctx-\u003edata and ctx-\u003edata_end\n    access based on ctx offset only. That caused loads in tracing programs\n    int bpf_prog(struct pt_regs *ctx) { .. ctx-\u003eax .. }\n    to be incorrectly marked as PTR_TO_PACKET which later caused verifier\n    to reject the program that was actually valid in tracing context.\n    Fix this by doing program type specific matching of ctx offsets.\n\n    Fixes: 969bf05eb3ce (\"bpf: direct packet access\")\n    Reported-by: Sasha Goldshtein \u003cgoldshtn@gmail.com\u003e\n    Signed-off-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Acked-by: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 3d870d75b08b0f778d81af9c1740cf33c7e32830\nAuthor: Arnaldo Carvalho de Melo \u003cacme@redhat.com\u003e\nDate:   Thu Apr 28 13:16:33 2016 -0300\n\n    perf core: Per event callchain limit\n\n    Additionally to being able to control the system wide maximum depth via\n    /proc/sys/kernel/perf_event_max_stack, now we are able to ask for\n    different depths per event, using perf_event_attr.sample_max_stack for\n    that.\n\n    This uses an u16 hole at the end of perf_event_attr, that, when\n    perf_event_attr.sample_type has the PERF_SAMPLE_CALLCHAIN, if\n    sample_max_stack is zero, means use perf_event_max_stack, otherwise\n    it\u0027ll be bounds checked under callchain_mutex.\n\n    Cc: Adrian Hunter \u003cadrian.hunter@intel.com\u003e\n    Cc: Alexander Shishkin \u003calexander.shishkin@linux.intel.com\u003e\n    Cc: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Cc: Brendan Gregg \u003cbrendan.d.gregg@gmail.com\u003e\n    Cc: David Ahern \u003cdsahern@gmail.com\u003e\n    Cc: Frederic Weisbecker \u003cfweisbec@gmail.com\u003e\n    Cc: He Kuang \u003chekuang@huawei.com\u003e\n    Cc: Jiri Olsa \u003cjolsa@redhat.com\u003e\n    Cc: Linus Torvalds \u003ctorvalds@linux-foundation.org\u003e\n    Cc: Masami Hiramatsu \u003cmhiramat@kernel.org\u003e\n    Cc: Milian Wolff \u003cmilian.wolff@kdab.com\u003e\n    Cc: Namhyung Kim \u003cnamhyung@kernel.org\u003e\n    Cc: Peter Zijlstra \u003cpeterz@infradead.org\u003e\n    Cc: Stephane Eranian \u003ceranian@google.com\u003e\n    Cc: Thomas Gleixner \u003ctglx@linutronix.de\u003e\n    Cc: Vince Weaver \u003cvincent.weaver@maine.edu\u003e\n    Cc: Wang Nan \u003cwangnan0@huawei.com\u003e\n    Cc: Zefan Li \u003clizefan@huawei.com\u003e\n    Link: http://lkml.kernel.org/n/tip-kolmn1yo40p7jhswxwrc7rrd@git.kernel.org\n    Signed-off-by: Arnaldo Carvalho de Melo \u003cacme@redhat.com\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit f856f9a7d03bf416794d969d319d763ae4a74166\nAuthor: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\nDate:   Sun May 22 23:16:18 2016 +0200\n\n    bpf, inode: disallow userns mounts\n\n    Follow-up to commit e27f4a942a0e (\"bpf: Use mount_nodev not mount_ns\n    to mount the bpf filesystem\"), which removes the FS_USERNS_MOUNT flag.\n\n    The original idea was to have a per mountns instance instead of a\n    single global fs instance, but that didn\u0027t work out and we had to\n    switch to mount_nodev() model. The intent of that middle ground was\n    that we avoid users who don\u0027t play nice to create endless instances\n    of bpf fs which are difficult to control and discover from an admin\n    point of view, but at the same time it would have allowed us to be\n    more flexible with regard to namespaces.\n\n    Therefore, since we now did the switch to mount_nodev() as a fix\n    where individual instances are created, we also need to remove userns\n    mount flag along with it to avoid running into mentioned situation.\n    I don\u0027t expect any breakage at this early point in time with removing\n    the flag and we can revisit this later should the requirement for\n    this come up with future users. This and commit e27f4a942a0e have\n    been split to facilitate tracking should any of them run into the\n    unlikely case of causing a regression.\n\n    Fixes: b2197755b263 (\"bpf: add support for persistent maps/progs\")\n    Signed-off-by: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Acked-by: Hannes Frederic Sowa \u003channes@stressinduktion.org\u003e\n    Acked-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit f875635e489ea6c4d7519d1bfddabc22c9a7cada\nAuthor: Alexei Starovoitov \u003cast@fb.com\u003e\nDate:   Thu May 19 18:17:14 2016 -0700\n\n    bpf: teach verifier to recognize imm +\u003d ptr pattern\n\n    Humans don\u0027t write C code like:\n      u8 *ptr \u003d skb-\u003edata;\n      int imm \u003d 4;\n      imm +\u003d ptr;\n    but from llvm backend point of view \u0027imm\u0027 and \u0027ptr\u0027 are registers and\n    imm +\u003d ptr may be preferred vs ptr +\u003d imm depending which register value\n    will be used further in the code, while verifier can only recognize ptr +\u003d imm.\n    That caused small unrelated changes in the C code of the bpf program to\n    trigger rejection by the verifier. Therefore teach the verifier to recognize\n    both ptr +\u003d imm and imm +\u003d ptr.\n    For example:\n    when R6\u003dpkt(id\u003d0,off\u003d0,r\u003d62) R7\u003dimm22\n    after r7 +\u003d r6 instruction\n    will be R6\u003dpkt(id\u003d0,off\u003d0,r\u003d62) R7\u003dpkt(id\u003d0,off\u003d22,r\u003d62)\n\n    Fixes: 969bf05eb3ce (\"bpf: direct packet access\")\n    Signed-off-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Acked-by: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 2181f10f7e5bd65c5828617f1c6cc977aac55736\nAuthor: Alexei Starovoitov \u003cast@fb.com\u003e\nDate:   Thu May 19 18:17:13 2016 -0700\n\n    bpf: support decreasing order in direct packet access\n\n    when packet headers are accessed in \u0027decreasing\u0027 order (like TCP port\n    may be fetched before the program reads IP src) the llvm may generate\n    the following code:\n    [...]                // R7\u003dpkt(id\u003d0,off\u003d22,r\u003d70)\n    r2 \u003d *(u32 *)(r7 +0) // good access\n    [...]\n    r7 +\u003d 40             // R7\u003dpkt(id\u003d0,off\u003d62,r\u003d70)\n    r8 \u003d *(u32 *)(r7 +0) // good access\n    [...]\n    r1 \u003d *(u32 *)(r7 -20) // this one will fail though it\u0027s within a safe range\n                          // it\u0027s doing *(u32*)(skb-\u003edata + 42)\n    Fix verifier to recognize such code pattern\n\n    Alos turned out that \u0027off \u003e range\u0027 condition is not a verifier bug.\n    It\u0027s a buggy program that may do something like:\n    if (ptr + 50 \u003e data_end)\n      return 0;\n    ptr +\u003d 60;\n    *(u32*)ptr;\n    in such case emit\n    \"invalid access to packet, off\u003d0 size\u003d4, R1(id\u003d0,off\u003d60,r\u003d50)\" error message,\n    so all information is available for the program author to fix the program.\n\n    Fixes: 969bf05eb3ce (\"bpf: direct packet access\")\n    Signed-off-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Acked-by: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit c29c7a5408970f3d7f223e6d949e3d6b0823c8a6\nAuthor: Eric W. Biederman \u003cebiederm@xmission.com\u003e\nDate:   Fri May 20 17:22:48 2016 -0500\n\n    bpf: Use mount_nodev not mount_ns to mount the bpf filesystem\n\n    While reviewing the filesystems that set FS_USERNS_MOUNT I spotted the\n    bpf filesystem.  Looking at the code I saw a broken usage of mount_ns\n    with current-\u003ensproxy-\u003emnt_ns. As the code does not acquire a\n    reference to the mount namespace it can not possibly be correct to\n    store the mount namespace on the superblock as it does.\n\n    Replace mount_ns with mount_nodev so that each mount of the bpf\n    filesystem returns a distinct instance, and the code is not buggy.\n\n    In discussion with Hannes Frederic Sowa it was reported that the use\n    of mount_ns was an attempt to have one bpf instance per mount\n    namespace, in an attempt to keep resources that pin resources from\n    hiding.  That intent simply does not work, the vfs is not built to\n    allow that kind of behavior.  Which means that the bpf filesystem\n    really is buggy both semantically and in it\u0027s implemenation as it does\n    not nor can it implement the original intent.\n\n    This change is userspace visible, but my experience with similar\n    filesystems leads me to believe nothing will break with a model of each\n    mount of the bpf filesystem is distinct from all others.\n\n    Fixes: b2197755b263 (\"bpf: add support for persistent maps/progs\")\n    Cc: Hannes Frederic Sowa \u003channes@stressinduktion.org\u003e\n    Acked-by: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Signed-off-by: \"Eric W. Biederman\" \u003cebiederm@xmission.com\u003e\n    Acked-by: Hannes Frederic Sowa \u003channes@stressinduktion.org\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 47617cdec8337fd1372a2da5b0eab6ba402e1d7f\nAuthor: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\nDate:   Wed May 18 14:14:28 2016 +0200\n\n    bpf: rather use get_random_int for randomizations\n\n    Start address randomization and blinding in BPF currently use\n    prandom_u32(). prandom_u32() values are not exposed to unpriviledged\n    user space to my knowledge, but given other kernel facilities such as\n    ASLR, stack canaries, etc make use of stronger get_random_int(), we\n    better make use of it here as well given blinding requests successively\n    new random values. get_random_int() has minimal entropy pool depletion,\n    is not cryptographically secure, but doesn\u0027t need to be for our use\n    cases here.\n\n    Suggested-by: Hannes Frederic Sowa \u003channes@stressinduktion.org\u003e\n    Signed-off-by: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Acked-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit d0c23ffd58bb1f3a07a17f305aa02fd0ea7314ca\nAuthor: Arnaldo Carvalho de Melo \u003cacme@redhat.com\u003e\nDate:   Thu Apr 28 12:30:53 2016 -0300\n\n    perf core: Pass max stack as a perf_callchain_entry context\n\n    This makes perf_callchain_{user,kernel}() receive the max stack\n    as context for the perf_callchain_entry, instead of accessing\n    the global sysctl_perf_event_max_stack.\n\n    Cc: Adrian Hunter \u003cadrian.hunter@intel.com\u003e\n    Cc: Alexander Shishkin \u003calexander.shishkin@linux.intel.com\u003e\n    Cc: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Cc: Brendan Gregg \u003cbrendan.d.gregg@gmail.com\u003e\n    Cc: David Ahern \u003cdsahern@gmail.com\u003e\n    Cc: Frederic Weisbecker \u003cfweisbec@gmail.com\u003e\n    Cc: He Kuang \u003chekuang@huawei.com\u003e\n    Cc: Jiri Olsa \u003cjolsa@redhat.com\u003e\n    Cc: Linus Torvalds \u003ctorvalds@linux-foundation.org\u003e\n    Cc: Masami Hiramatsu \u003cmhiramat@kernel.org\u003e\n    Cc: Milian Wolff \u003cmilian.wolff@kdab.com\u003e\n    Cc: Namhyung Kim \u003cnamhyung@kernel.org\u003e\n    Cc: Peter Zijlstra \u003cpeterz@infradead.org\u003e\n    Cc: Stephane Eranian \u003ceranian@google.com\u003e\n    Cc: Thomas Gleixner \u003ctglx@linutronix.de\u003e\n    Cc: Vince Weaver \u003cvincent.weaver@maine.edu\u003e\n    Cc: Wang Nan \u003cwangnan0@huawei.com\u003e\n    Cc: Zefan Li \u003clizefan@huawei.com\u003e\n    Link: http://lkml.kernel.org/n/tip-kolmn1yo40p7jhswxwrc7rrd@git.kernel.org\n    Signed-off-by: Arnaldo Carvalho de Melo \u003cacme@redhat.com\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 1078c71744247854c7c4113949c9f1d30c830209\nAuthor: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\nDate:   Fri May 13 19:08:32 2016 +0200\n\n    bpf: add generic constant blinding for use in jits\n\n    This work adds a generic facility for use from eBPF JIT compilers\n    that allows for further hardening of JIT generated images through\n    blinding constants. In response to the original work on BPF JIT\n    spraying published by Keegan McAllister [1], most BPF JITs were\n    changed to make images read-only and start at a randomized offset\n    in the page, where the rest was filled with trap instructions. We\n    have this nowadays in x86, arm, arm64 and s390 JIT compilers.\n    Additionally, later work also made eBPF interpreter images read\n    only for kernels supporting DEBUG_SET_MODULE_RONX, that is, x86,\n    arm, arm64 and s390 archs as well currently. This is done by\n    default for mentioned JITs when JITing is enabled. Furthermore,\n    we had a generic and configurable constant blinding facility on our\n    todo for quite some time now to further make spraying harder, and\n    first implementation since around netconf 2016.\n\n    We found that for systems where untrusted users can load cBPF/eBPF\n    code where JIT is enabled, start offset randomization helps a bit\n    to make jumps into crafted payload harder, but in case where larger\n    programs that cross page boundary are injected, we again have some\n    part of the program opcodes at a page start offset. With improved\n    guessing and more reliable payload injection, chances can increase\n    to jump into such payload. Elena Reshetova recently wrote a test\n    case for it [2, 3]. Moreover, eBPF comes with 64 bit constants, which\n    can leave some more room for payloads. Note that for all this,\n    additional bugs in the kernel are still required to make the jump\n    (and of course to guess right, to not jump into a trap) and naturally\n    the JIT must be enabled, which is disabled by default.\n\n    For helping mitigation, the general idea is to provide an option\n    bpf_jit_harden that admins can tweak along with bpf_jit_enable, so\n    that for cases where JIT should be enabled for performance reasons,\n    the generated image can be further hardened with blinding constants\n    for unpriviledged users (bpf_jit_harden \u003d\u003d 1), with trading off\n    performance for these, but not for privileged ones. We also added\n    the option of blinding for all users (bpf_jit_harden \u003d\u003d 2), which\n    is quite helpful for testing f.e. with test_bpf.ko. There are no\n    further e.g. hardening levels of bpf_jit_harden switch intended,\n    rationale is to have it dead simple to use as on/off. Since this\n    functionality would need to be duplicated over and over for JIT\n    compilers to use, which are already complex enough, we provide a\n    generic eBPF byte-code level based blinding implementation, which is\n    then just transparently JITed. JIT compilers need to make only a few\n    changes to integrate this facility and can be migrated one by one.\n\n    This option is for eBPF JITs and will be used in x86, arm64, s390\n    without too much effort, and soon ppc64 JITs, thus that native eBPF\n    can be blinded as well as cBPF to eBPF migrations, so that both can\n    be covered with a single implementation. The rule for JITs is that\n    bpf_jit_blind_constants() must be called from bpf_int_jit_compile(),\n    and in case blinding is disabled, we follow normally with JITing the\n    passed program. In case blinding is enabled and we fail during the\n    process of blinding itself, we must return with the interpreter.\n    Similarly, in case the JITing process after the blinding failed, we\n    return normally to the interpreter with the non-blinded code. Meaning,\n    interpreter doesn\u0027t change in any way and operates on eBPF code as\n    usual. For doing this pre-JIT blinding step, we need to make use of\n    a helper/auxiliary register, here BPF_REG_AX. This is strictly internal\n    to the JIT and not in any way part of the eBPF architecture. Just like\n    in the same way as JITs internally make use of some helper registers\n    when emitting code, only that here the helper register is one\n    abstraction level higher in eBPF bytecode, but nevertheless in JIT\n    phase. That helper register is needed since f.e. manually written\n    program can issue loads to all registers of eBPF architecture.\n\n    The core concept with the additional register is: blind out all 32\n    and 64 bit constants by converting BPF_K based instructions into a\n    small sequence from K_VAL into ((RND ^ K_VAL) ^ RND). Therefore, this\n    is transformed into: BPF_REG_AX :\u003d (RND ^ K_VAL), BPF_REG_AX ^\u003d RND,\n    and REG \u003cOP\u003e BPF_REG_AX, so actual operation on the target register\n    is translated from BPF_K into BPF_X one that is operating on\n    BPF_REG_AX\u0027s content. During rewriting phase when blinding, RND is\n    newly generated via prandom_u32() for each processed instruction.\n    64 bit loads are split into two 32 bit loads to make translation and\n    patching not too complex. Only basic thing required by JITs is to\n    call the helper bpf_jit_blind_constants()/bpf_jit_prog_release_other()\n    pair, and to map BPF_REG_AX into an unused register.\n\n    Small bpf_jit_disasm extract from [2] when applied to x86 JIT:\n\n    echo 0 \u003e /proc/sys/net/core/bpf_jit_harden\n\n      ffffffffa034f5e9 + \u003cx\u003e:\n      [...]\n      39:   mov    $0xa8909090,%eax\n      3e:   mov    $0xa8909090,%eax\n      43:   mov    $0xa8ff3148,%eax\n      48:   mov    $0xa89081b4,%eax\n      4d:   mov    $0xa8900bb0,%eax\n      52:   mov    $0xa810e0c1,%eax\n      57:   mov    $0xa8908eb4,%eax\n      5c:   mov    $0xa89020b0,%eax\n      [...]\n\n    echo 1 \u003e /proc/sys/net/core/bpf_jit_harden\n\n      ffffffffa034f1e5 + \u003cx\u003e:\n      [...]\n      39:   mov    $0xe1192563,%r10d\n      3f:   xor    $0x4989b5f3,%r10d\n      46:   mov    %r10d,%eax\n      49:   mov    $0xb8296d93,%r10d\n      4f:   xor    $0x10b9fd03,%r10d\n      56:   mov    %r10d,%eax\n      59:   mov    $0x8c381146,%r10d\n      5f:   xor    $0x24c7200e,%r10d\n      66:   mov    %r10d,%eax\n      69:   mov    $0xeb2a830e,%r10d\n      6f:   xor    $0x43ba02ba,%r10d\n      76:   mov    %r10d,%eax\n      79:   mov    $0xd9730af,%r10d\n      7f:   xor    $0xa5073b1f,%r10d\n      86:   mov    %r10d,%eax\n      89:   mov    $0x9a45662b,%r10d\n      8f:   xor    $0x325586ea,%r10d\n      96:   mov    %r10d,%eax\n      [...]\n\n    As can be seen, original constants that carry payload are hidden\n    when enabled, actual operations are transformed from constant-based\n    to register-based ones, making jumps into constants ineffective.\n    Above extract/example uses single BPF load instruction over and\n    over, but of course all instructions with constants are blinded.\n\n    Performance wise, JIT with blinding performs a bit slower than just\n    JIT and faster than interpreter case. This is expected, since we\n    still get all the performance benefits from JITing and in normal\n    use-cases not every single instruction needs to be blinded. Summing\n    up all 296 test cases averaged over multiple runs from test_bpf.ko\n    suite, interpreter was 55% slower than JIT only and JIT with blinding\n    was 8% slower than JIT only. Since there are also some extremes in\n    the test suite, I expect for ordinary workloads that the performance\n    for the JIT with blinding case is even closer to JIT only case,\n    f.e. nmap test case from suite has averaged timings in ns 29 (JIT),\n    35 (+ blinding), and 151 (interpreter).\n\n    BPF test suite, seccomp test suite, eBPF sample code and various\n    bigger networking eBPF programs have been tested with this and were\n    running fine. For testing purposes, I also adapted interpreter and\n    redirected blinded eBPF image to interpreter and also here all tests\n    pass.\n\n      [1] http://mainisusuallyafunction.blogspot.com/2012/11/attacking-hardened-linux-systems-with.html\n      [2] https://github.com/01org/jit-spray-poc-for-ksp/\n      [3] http://www.openwall.com/lists/kernel-hardening/2016/05/03/5\n\n    Signed-off-by: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Reviewed-by: Elena Reshetova \u003celena.reshetova@intel.com\u003e\n    Acked-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit aafe2056243377c1af840c886a914d39e9e8c6db\nAuthor: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\nDate:   Fri May 13 19:08:27 2016 +0200\n\n    bpf: move bpf_jit_enable declaration\n\n    Move the bpf_jit_enable declaration to the filter.h file where\n    most other core code is declared, also since we\u0027re going to add\n    a second knob there.\n\n    Signed-off-by: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Acked-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit b11b43db4b3336724bb40ac5c52a292862f65b65\nAuthor: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\nDate:   Sat Jun 4 20:50:59 2016 +0200\n\n    bpf, trace: use READ_ONCE for retrieving file ptr\n\n    In bpf_perf_event_read() and bpf_perf_event_output(), we must use\n    READ_ONCE() for fetching the struct file pointer, which could get\n    updated concurrently, so we must prevent the compiler from potential\n    refetching.\n\n    We already do this with tail calls for fetching the related bpf_prog,\n    but not so on stored perf events. Semantics for both are the same\n    with regards to updates.\n\n    Fixes: a43eec304259 (\"bpf: introduce bpf_perf_event_output() helper\")\n    Fixes: 35578d798400 (\"bpf: Implement function bpf_perf_event_read() that get the selected hardware PMU conuter\")\n    Signed-off-by: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Acked-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 7833af0f20ddc7ce7f7ce93aa7e1baecd4afb0fc\nAuthor: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\nDate:   Mon Apr 18 21:01:23 2016 +0200\n\n    bpf, trace: add BPF_F_CURRENT_CPU flag for bpf_perf_event_output\n\n    Add a BPF_F_CURRENT_CPU flag to optimize the use-case where user space has\n    per-CPU ring buffers and the eBPF program pushes the data into the current\n    CPU\u0027s ring buffer which saves us an extra helper function call in eBPF.\n    Also, make sure to properly reserve the remaining flags which are not used.\n\n    Signed-off-by: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Signed-off-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 2322bf96f733600232f2c4278b130e1d7c19f491\nAuthor: Arnd Bergmann \u003carnd@arndb.de\u003e\nDate:   Sat Apr 16 22:29:33 2016 +0200\n\n    bpf: avoid warning for wrong pointer cast\n\n    Two new functions in bpf contain a cast from a \u0027u64\u0027 to a\n    pointer. This works on 64-bit architectures but causes a warning\n    on all 32-bit architectures:\n\n    kernel/trace/bpf_trace.c: In function \u0027bpf_perf_event_output_tp\u0027:\n    kernel/trace/bpf_trace.c:350:13: error: cast to pointer from integer of different size [-Werror\u003dint-to-pointer-cast]\n      u64 ctx \u003d *(long *)r1;\n\n    This changes the cast to first convert the u64 argument into a uintptr_t,\n    which is guaranteed to be the same size as a pointer.\n\n    Signed-off-by: Arnd Bergmann \u003carnd@arndb.de\u003e\n    Fixes: 9940d67c93b5 (\"bpf: support bpf_get_stackid() and bpf_perf_event_output() in tracepoint programs\")\n    Acked-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 5349ef5f720bc92e473ccde5ca49b2ee4ba878cb\nAuthor: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\nDate:   Fri May 13 19:08:31 2016 +0200\n\n    bpf: prepare bpf_int_jit_compile/bpf_prog_select_runtime apis\n\n    Since the blinding is strictly only called from inside eBPF JITs,\n    we need to change signatures for bpf_int_jit_compile() and\n    bpf_prog_select_runtime() first in order to prepare that the\n    eBPF program we\u0027re dealing with can change underneath. Hence,\n    for call sites, we need to return the latest prog. No functional\n    change in this patch.\n\n    Signed-off-by: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Acked-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 192e276c6a5071da00428fc6b5cde067c5c01702\nAuthor: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\nDate:   Fri May 13 19:08:30 2016 +0200\n\n    bpf: add bpf_patch_insn_single helper\n\n    Move the functionality to patch instructions out of the verifier\n    code and into the core as the new bpf_patch_insn_single() helper\n    will be needed later on for blinding as well. No changes in\n    functionality.\n\n    Signed-off-by: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Acked-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit dcd52ad90b5721da28528f807e77dd0dc3d7962f\nAuthor: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\nDate:   Fri May 13 19:08:26 2016 +0200\n\n    bpf: minor cleanups in ebpf code\n\n    Besides others, remove redundant comments where the code is self\n    documenting enough, and properly indent various bpf_verifier_ops\n    and bpf_prog_type_list declarations. Moreover, remove two exports\n    that actually have no module user.\n\n    Signed-off-by: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Acked-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit d4e7499c984128d8509f5ac3bf647f3295d84c4a\nAuthor: Alexei Starovoitov \u003cast@fb.com\u003e\nDate:   Thu May 5 19:49:11 2016 -0700\n\n    bpf: improve verifier state equivalence\n\n    since UNKNOWN_VALUE type is weaker than CONST_IMM we can un-teach\n    verifier its recognition of constants in conditional branches\n    without affecting safety.\n    Ex:\n    if (reg \u003d\u003d 123) {\n      .. here verifier was marking reg-\u003etype as CONST_IMM\n         instead keep reg as UNKNOWN_VALUE\n    }\n\n    Two verifier states with UNKNOWN_VALUE are equivalent, whereas\n    CONST_IMM_X !\u003d CONST_IMM_Y, since CONST_IMM is used for stack range\n    verification and other cases.\n    So help search pruning by marking registers as UNKNOWN_VALUE\n    where possible instead of CONST_IMM.\n\n    Signed-off-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Acked-by: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit a74ad8ba8c5203757e5c0b7d445dc7224df4b18e\nAuthor: Alexei Starovoitov \u003cast@fb.com\u003e\nDate:   Thu May 5 19:49:10 2016 -0700\n\n    bpf: direct packet access\n\n    Extended BPF carried over two instructions from classic to access\n    packet data: LD_ABS and LD_IND. They\u0027re highly optimized in JITs,\n    but due to their design they have to do length check for every access.\n    When BPF is processing 20M packets per second single LD_ABS after JIT\n    is consuming 3% cpu. Hence the need to optimize it further by amortizing\n    the cost of \u0027off \u003c skb_headlen\u0027 over multiple packet accesses.\n    One option is to introduce two new eBPF instructions LD_ABS_DW and LD_IND_DW\n    with similar usage as skb_header_pointer().\n    The kernel part for interpreter and x64 JIT was implemented in [1], but such\n    new insns behave like old ld_abs and abort the program with \u0027return 0\u0027 if\n    access is beyond linear data. Such hidden control flow is hard to workaround\n    plus changing JITs and rolling out new llvm is incovenient.\n\n    Therefore allow cls_bpf/act_bpf program access skb-\u003edata directly:\n    int bpf_prog(struct __sk_buff *skb)\n    {\n      struct iphdr *ip;\n\n      if (skb-\u003edata + sizeof(struct iphdr) + ETH_HLEN \u003e skb-\u003edata_end)\n          /* packet too small */\n          return 0;\n\n      ip \u003d skb-\u003edata + ETH_HLEN;\n\n      /* access IP header fields with direct loads */\n      if (ip-\u003eversion !\u003d 4 || ip-\u003esaddr \u003d\u003d 0x7f000001)\n          return 1;\n      [...]\n    }\n\n    This solution avoids introduction of new instructions. llvm stays\n    the same and all JITs stay the same, but verifier has to work extra hard\n    to prove safety of the above program.\n\n    For XDP the direct store instructions can be allowed as well.\n\n    The skb-\u003edata is NET_IP_ALIGNED, so for common cases the verifier can check\n    the alignment. The complex packet parsers where packet pointer is adjusted\n    incrementally cannot be tracked for alignment, so allow byte access in such cases\n    and misaligned access on architectures that define efficient_unaligned_access\n\n    [1] https://git.kernel.org/cgit/linux/kernel/git/ast/bpf.git/?h\u003dld_abs_dw\n\n    Signed-off-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Acked-by: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 80227b6f68c8f8b482fd91a972b3e63bd51f3f2d\nAuthor: Alexei Starovoitov \u003cast@fb.com\u003e\nDate:   Thu May 5 19:49:09 2016 -0700\n\n    bpf: cleanup verifier code\n\n    cleanup verifier code and prepare it for addition of \"pointer to packet\" logic\n\n    Signed-off-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Acked-by: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 7d51802990f19293d3019010dab50455612c3ddd\nAuthor: Alexei Starovoitov \u003cast@fb.com\u003e\nDate:   Wed Apr 27 18:56:21 2016 -0700\n\n    bpf: fix check_map_func_compatibility logic\n\n    The commit 35578d798400 (\"bpf: Implement function bpf_perf_event_read() that get the selected hardware PMU conuter\")\n    introduced clever way to check bpf_helper\u003c-\u003emap_type compatibility.\n    Later on commit a43eec304259 (\"bpf: introduce bpf_perf_event_output() helper\") adjusted\n    the logic and inadvertently broke it.\n    Get rid of the clever bool compare and go back to two-way check\n    from map and from helper perspective.\n\n    Fixes: a43eec304259 (\"bpf: introduce bpf_perf_event_output() helper\")\n    Reported-by: Jann Horn \u003cjannh@google.com\u003e\n    Signed-off-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Signed-off-by: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 58931fc4de12dd0b23f0b0bc8609b2bdbf519757\nAuthor: Alexei Starovoitov \u003cast@fb.com\u003e\nDate:   Wed Apr 27 18:56:20 2016 -0700\n\n    bpf: fix refcnt overflow\n\n    On a system with \u003e32Gbyte of phyiscal memory and infinite RLIMIT_MEMLOCK,\n    the malicious application may overflow 32-bit bpf program refcnt.\n    It\u0027s also possible to overflow map refcnt on 1Tb system.\n    Impose 32k hard limit which means that the same bpf program or\n    map cannot be shared by more than 32k processes.\n\n    Fixes: 1be7f75d1668 (\"bpf: enable non-root eBPF programs\")\n    Reported-by: Jann Horn \u003cjannh@google.com\u003e\n    Signed-off-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Acked-by: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 3f4206f8f81aa8a5c0f7e96ea8fd656bdddb524e\nAuthor: Arnaldo Carvalho de Melo \u003cacme@redhat.com\u003e\nDate:   Thu Apr 21 12:28:50 2016 -0300\n\n    perf core: Allow setting up max frame stack depth via sysctl\n\n    The default remains 127, which is good for most cases, and not even hit\n    most of the time, but then for some cases, as reported by Brendan, 1024+\n    deep frames are appearing on the radar for things like groovy, ruby.\n\n    And in some workloads putting a _lower_ cap on this may make sense. One\n    that is per event still needs to be put in place tho.\n\n    The new file is:\n\n      # cat /proc/sys/kernel/perf_event_max_stack\n      127\n\n    Chaging it:\n\n      # echo 256 \u003e /proc/sys/kernel/perf_event_max_stack\n      # cat /proc/sys/kernel/perf_event_max_stack\n      256\n\n    But as soon as there is some event using callchains we get:\n\n      # echo 512 \u003e /proc/sys/kernel/perf_event_max_stack\n      -bash: echo: write error: Device or resource busy\n      #\n\n    Because we only allocate the callchain percpu data structures when there\n    is a user, which allows for changing the max easily, its just a matter\n    of having no callchain users at that point.\n\n    Reported-and-Tested-by: Brendan Gregg \u003cbrendan.d.gregg@gmail.com\u003e\n    Reviewed-by: Frederic Weisbecker \u003cfweisbec@gmail.com\u003e\n    Acked-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Acked-by: David Ahern \u003cdsahern@gmail.com\u003e\n    Cc: Adrian Hunter \u003cadrian.hunter@intel.com\u003e\n    Cc: Alexander Shishkin \u003calexander.shishkin@linux.intel.com\u003e\n    Cc: He Kuang \u003chekuang@huawei.com\u003e\n    Cc: Jiri Olsa \u003cjolsa@redhat.com\u003e\n    Cc: Linus Torvalds \u003ctorvalds@linux-foundation.org\u003e\n    Cc: Masami Hiramatsu \u003cmhiramat@kernel.org\u003e\n    Cc: Milian Wolff \u003cmilian.wolff@kdab.com\u003e\n    Cc: Namhyung Kim \u003cnamhyung@kernel.org\u003e\n    Cc: Peter Zijlstra \u003cpeterz@infradead.org\u003e\n    Cc: Stephane Eranian \u003ceranian@google.com\u003e\n    Cc: Thomas Gleixner \u003ctglx@linutronix.de\u003e\n    Cc: Vince Weaver \u003cvincent.weaver@maine.edu\u003e\n    Cc: Wang Nan \u003cwangnan0@huawei.com\u003e\n    Cc: Zefan Li \u003clizefan@huawei.com\u003e\n    Link: http://lkml.kernel.org/r/20160426002928.GB16708@kernel.org\n    Signed-off-by: Arnaldo Carvalho de Melo \u003cacme@redhat.com\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 055afa297fdfd4beb890a1cbd8428cd01e9953c5\nAuthor: Alexei Starovoitov \u003cast@fb.com\u003e\nDate:   Wed Feb 17 19:58:57 2016 -0800\n\n    perf: generalize perf_callchain\n\n    . avoid walking the stack when there is no room left in the buffer\n    . generalize get_perf_callchain() to be called from bpf helper\n\n    Signed-off-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit a3ea6d47cc0eb62ce408ed6d4353abc5739266f4\nAuthor: Jann Horn \u003cjannh@google.com\u003e\nDate:   Tue Apr 26 22:26:26 2016 +0200\n\n    bpf: fix double-fdput in replace_map_fd_with_map_ptr()\n\n    When bpf(BPF_PROG_LOAD, ...) was invoked with a BPF program whose bytecode\n    references a non-map file descriptor as a map file descriptor, the error\n    handling code called fdput() twice instead of once (in __bpf_map_get() and\n    in replace_map_fd_with_map_ptr()). If the file descriptor table of the\n    current task is shared, this causes f_count to be decremented too much,\n    allowing the struct file to be freed while it is still in use\n    (use-after-free). This can be exploited to gain root privileges by an\n    unprivileged user.\n\n    This bug was introduced in\n    commit 0246e64d9a5f (\"bpf: handle pseudo BPF_LD_IMM64 insn\"), but is only\n    exploitable since\n    commit 1be7f75d1668 (\"bpf: enable non-root eBPF programs\") because\n    previously, CAP_SYS_ADMIN was required to reach the vulnerable code.\n\n    (posted publicly according to request by maintainer)\n\n    Signed-off-by: Jann Horn \u003cjannh@google.com\u003e\n    Signed-off-by: Linus Torvalds \u003ctorvalds@linux-foundation.org\u003e\n    Acked-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Acked-by: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 4a8ab235c7c50b3000e388aa86738d16a1cf9224\nAuthor: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\nDate:   Mon Apr 18 21:01:24 2016 +0200\n\n    bpf: add event output helper for notifications/sampling/logging\n\n    This patch adds a new helper for cls/act programs that can push events\n    to user space applications. For networking, this can be f.e. for sampling,\n    debugging, logging purposes or pushing of arbitrary wake-up events. The\n    idea is similar to a43eec304259 (\"bpf: introduce bpf_perf_event_output()\n    helper\") and 39111695b1b8 (\"samples: bpf: add bpf_perf_event_output example\").\n\n    The eBPF program utilizes a perf event array map that user space populates\n    with fds from perf_event_open(), the eBPF program calls into the helper\n    f.e. as skb_event_output(skb, \u0026my_map, BPF_F_CURRENT_CPU, raw, sizeof(raw))\n    so that the raw data is pushed into the fd f.e. at the map index of the\n    current CPU.\n\n    User space can poll/mmap/etc on this and has a data channel for receiving\n    events that can be post-processed. The nice thing is that since the eBPF\n    program and user space application making use of it are tightly coupled,\n    they can define their own arbitrary raw data format and what/when they\n    want to push.\n\n    While f.e. packet headers could be one part of the meta data that is being\n    pushed, this is not a substitute for things like packet sockets as whole\n    packet is not being pushed and push is only done in a single direction.\n    Intention is more of a generically usable, efficient event pipe to applications.\n    Workflow is that tc can pin the map and applications can attach themselves\n    e.g. after cls/act setup to one or multiple map slots, demuxing is done by\n    the eBPF program.\n\n    Adding this facility is with minimal effort, it reuses the helper\n    introduced in a43eec304259 (\"bpf: introduce bpf_perf_event_output() helper\")\n    and we get its functionality for free by overloading its BPF_FUNC_ identifier\n    for cls/act programs, ctx is currently unused, but will be made use of in\n    future. Example will be added to iproute2\u0027s BPF example files.\n\n    Signed-off-by: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Signed-off-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 1e65fa3c7d3cddc9e164ccf92b7bc2f58af295ec\nAuthor: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\nDate:   Wed Apr 13 00:10:52 2016 +0200\n\n    bpf: convert relevant helper args to ARG_PTR_TO_RAW_STACK\n\n    This patch converts all helpers that can use ARG_PTR_TO_RAW_STACK as argument\n    type. For tc programs this is bpf_skb_load_bytes(), bpf_skb_get_tunnel_key(),\n    bpf_skb_get_tunnel_opt(). For tracing, this optimizes bpf_get_current_comm()\n    and bpf_probe_read(). The check in bpf_skb_load_bytes() for MAX_BPF_STACK can\n    also be removed since the verifier already makes sure we stay within bounds\n    on stack buffers.\n\n    Signed-off-by: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Acked-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit ae0139946a0417c15215f615e076d2435cc14259\nAuthor: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\nDate:   Wed Mar 30 00:02:00 2016 +0200\n\n    bpf: make padding in bpf_tunnel_key explicit\n\n    Make the 2 byte padding in struct bpf_tunnel_key between tunnel_ttl\n    and tunnel_label members explicit. No issue has been observed, and\n    gcc/llvm does padding for the old struct already, where tunnel_label\n    was not yet present, so the current code works, but since it\u0027s part\n    of uapi, make sure we don\u0027t introduce holes in structs.\n\n    Therefore, add tunnel_ext that we can use generically in future\n    (f.e. to flag OAM messages for backends, etc). Also add the offset\n    to the compat tests to be sure should some compilers not padd the\n    tail of the old version of bpf_tunnel_key.\n\n    Fixes: 4018ab1875e0 (\"bpf: support flow label for bpf_skb_{set, get}_tunnel_key\")\n    Signed-off-by: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Acked-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 3bcd51201b2497fa46a6bad2b0b0ddd47ae7061a\nAuthor: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\nDate:   Wed Mar 16 01:42:51 2016 +0100\n\n    ip_tunnels, bpf: define IP_TUNNEL_OPTS_MAX and use it\n\n    eBPF defines this as BPF_TUNLEN_MAX and OVS just uses the hard-coded\n    value inside struct sw_flow_key. Thus, add and use IP_TUNNEL_OPTS_MAX\n    for this, which makes the code a bit more generic and allows to remove\n    BPF_TUNLEN_MAX from eBPF code.\n\n    Signed-off-by: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit b20ac310bea9199559b104448f38c9c3a088d5b5\nAuthor: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\nDate:   Wed Mar 16 01:42:50 2016 +0100\n\n    bpf, dst: add and use dst_tclassid helper\n\n    We can just add a small helper dst_tclassid() for retrieving the\n    dst-\u003etclassid value. It makes the code a bit better in that we can\n    get rid of the ifdef from filter.c by moving this into the header.\n\n    Signed-off-by: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 3f13d80149ee747b142959233b94fb5ba0b21f23\nAuthor: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\nDate:   Wed Mar 16 01:42:49 2016 +0100\n\n    bpf: make skb-\u003etc_classid also readable\n\n    Currently, the tc_classid from eBPF skb context is write-only, but there\u0027s\n    no good reason for tc programs to limit it to write-only. For example,\n    it can be used to transfer its state via tail calls where the resulting\n    tc_classid gets filled gradually.\n\n    Signed-off-by: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Acked-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit b3b2fa8199462acadf80fdcbfba82bb5d480a76f\nAuthor: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\nDate:   Wed Mar 9 03:00:05 2016 +0100\n\n    bpf: support flow label for bpf_skb_{set, get}_tunnel_key\n\n    This patch extends bpf_tunnel_key with a tunnel_label member, that maps\n    to ip_tunnel_key\u0027s label so underlying backends like vxlan and geneve\n    can propagate the label to udp_tunnel6_xmit_skb(), where it\u0027s being set\n    in the IPv6 header. It allows for having 20 more bits to encode/decode\n    flow related meta information programmatically. Tested with vxlan and\n    geneve.\n\n    Signed-off-by: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Acked-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 802357ddae788cdbcd3ad37bddaa1bb5bc68c36c\nAuthor: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\nDate:   Fri Mar 4 15:15:06 2016 +0100\n\n    bpf: support for access to tunnel options\n\n    After eBPF being able to programmatically access/manage tunnel key meta\n    data via commit d3aa45ce6b94 (\"bpf: add helpers to access tunnel metadata\")\n    and more recently also for IPv6 through c6c33454072f (\"bpf: support ipv6\n    for bpf_skb_{set,get}_tunnel_key\"), this work adds two complementary\n    helpers to generically access their auxiliary tunnel options.\n\n    Geneve and vxlan support this facility. For geneve, TLVs can be pushed,\n    and for the vxlan case its GBP extension. I.e. setting tunnel key for geneve\n    case only makes sense, if we can also read/write TLVs into it. In the GBP\n    case, it provides the flexibility to easily map the group policy ID in\n    combination with other helpers or maps.\n\n    I chose to model this as two separate helpers, bpf_skb_{set,get}_tunnel_opt(),\n    for a couple of reasons. bpf_skb_{set,get}_tunnel_key() is already rather\n    complex by itself, and there may be cases for tunnel key backends where\n    tunnel options are not always needed. If we would have integrated this\n    into bpf_skb_{set,get}_tunnel_key() nevertheless, we are very limited with\n    remaining helper arguments, so keeping compatibility on structs in case of\n    passing in a flat buffer gets more cumbersome. Separating both also allows\n    for more flexibility and future extensibility, f.e. options could be fed\n    directly from a map, etc.\n\n    Moreover, change geneve\u0027s xmit path to test only for info-\u003eoptions_len\n    instead of TUNNEL_GENEVE_OPT flag. This makes it more consistent with vxlan\u0027s\n    xmit path and allows for avoiding to specify a protocol flag in the API on\n    xmit, so it can be protocol agnostic. Having info-\u003eoptions_len is enough\n    information that is needed. Tested with vxlan and geneve.\n\n    Signed-off-by: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Acked-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit fe0fb5c15b4f9450ced349b014bbb959fdf061db\nAuthor: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\nDate:   Fri Mar 4 15:15:05 2016 +0100\n\n    bpf: allow to propagate df in bpf_skb_set_tunnel_key\n\n    Added by 9a628224a61b (\"ip_tunnel: Add dont fragment flag.\"), allow to\n    feed df flag into tunneling facilities (currently supported on TX by\n    vxlan, geneve and gre) as a hint from eBPF\u0027s bpf_skb_set_tunnel_key()\n    helper.\n\n    Signed-off-by: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Acked-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 6d03106a7d802d3c7bf9db8663855e2e546827d4\nAuthor: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\nDate:   Fri Mar 4 15:15:04 2016 +0100\n\n    bpf: make helper function protos static\n\n    They are only used here, so there\u0027s no reason they should not be static.\n    Only the vlan push/pop protos are used in the test_bpf suite.\n\n    Signed-off-by: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Acked-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 626cceff94ac482ea00425dbfc264911a122215d\nAuthor: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\nDate:   Fri Mar 4 15:15:03 2016 +0100\n\n    bpf: add flags to bpf_skb_store_bytes for clearing hash\n\n    When overwriting parts of the packet with bpf_skb_store_bytes() that\n    were fed previously into skb-\u003ehash calculation, we should clear the\n    current hash with skb_clear_hash(), so that a next skb_get_hash() call\n    can determine the correct hash related to this skb.\n\n    Signed-off-by: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Acked-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit fdf2f59cd69829c41498e9ec68dd56d0faf15e27\nAuthor: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\nDate:   Fri Mar 4 15:15:02 2016 +0100\n\n    bpf: allow bpf_csum_diff to feed bpf_l3_csum_replace as well\n\n    Commit 7d672345ed29 (\"bpf: add generic bpf_csum_diff helper\") added a\n    generic checksum diff helper that can feed bpf_l4_csum_replace() with\n    a target __wsum diff that is to be applied to the L4 checksum. This\n    facility is very flexible, can be cascaded, allows for adding, removing,\n    or diffing data, or for calculating the pseudo header checksum from\n    scratch, but it can also be reused for working with the IPv4 header\n    checksum.\n\n    Thus, analogous to bpf_l4_csum_replace(), add a case for header field\n    value of 0 to change the checksum at a given offset through a new helper\n    csum_replace_by_diff(). Also, in addition to that, this provides an\n    easy to use interface for feeding precalculated diffs f.e. coming from\n    a map. It nicely complements bpf_l3_csum_replace() that currently allows\n    only for csum updates of 2 and 4 byte diffs.\n\n    Signed-off-by: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Acked-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 648d4535f4b232f73d4eed3e3bd24a30d4ab4462\nAuthor: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\nDate:   Tue Feb 23 02:05:26 2016 +0100\n\n    bpf: fix csum setting for bpf_set_tunnel_key\n\n    The fix in 35e2d1152b22 (\"tunnels: Allow IPv6 UDP checksums to be correctly\n    controlled.\") changed behavior for bpf_set_tunnel_key() when in use with\n    IPv6 and thus uncovered a bug that TUNNEL_CSUM needed to be set but wasn\u0027t.\n    As a result, the stack dropped ingress vxlan IPv6 packets, that have been\n    sent via eBPF through collect meta data mode due to checksum now being zero.\n\n    Since after LCO, we enable IPv4 checksum by default, so make that analogous\n    and only provide a flag BPF_F_ZERO_CSUM_TX for the user to turn it off in\n    IPv4 case.\n\n    Fixes: 35e2d1152b22 (\"tunnels: Allow IPv6 UDP checksums to be correctly controlled.\")\n    Fixes: c6c33454072f (\"bpf: support ipv6 for bpf_skb_{set,get}_tunnel_key\")\n    Signed-off-by: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Acked-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit ed44455580e72cc8c9f31cbc3c5d10cc4899302e\nAuthor: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\nDate:   Fri Feb 19 23:05:26 2016 +0100\n\n    bpf: fix csum update in bpf_l4_csum_replace helper for udp\n\n    When using this helper for updating UDP checksums, we need to extend\n    this in order to write CSUM_MANGLED_0 for csum computations that result\n    into 0 as sum. Reason we need this is because packets with a checksum\n    could otherwise become incorrectly marked as a packet without a checksum.\n    Likewise, if the user indicates BPF_F_MARK_MANGLED_0, then we should\n    not turn packets without a checksum into ones with a checksum.\n\n    Signed-off-by: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Acked-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit aa4e2e4ae0d05d08035d8d4b64e7a8f3cacb7c06\nAuthor: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\nDate:   Fri Feb 19 23:05:27 2016 +0100\n\n    bpf: don\u0027t emit mov A,A on return\n\n    While debugging with bpf_jit_disasm I noticed emissions of \u0027mov %eax,%eax\u0027,\n    and found that this comes from BPF_RET | BPF_A translations from classic\n    BPF. Emitting this is unnecessary as BPF_REG_A is mapped into BPF_REG_0\n    already, therefore only emit a mov when immediates are used as return value.\n\n    Signed-off-by: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Acked-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 82d73e7a1eeec0d35c441677415876f2a049fadb\nAuthor: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\nDate:   Fri Feb 19 23:05:24 2016 +0100\n\n    bpf: remove artificial bpf_skb_{load, store}_bytes buffer limitation\n\n    We currently limit bpf_skb_store_bytes() and bpf_skb_load_bytes()\n    helpers to only store or load a maximum buffer of 16 bytes. Thus,\n    loading, rewriting and storing headers require several bpf_skb_load_bytes()\n    and bpf_skb_store_bytes() calls.\n\n    Also here we can use a per-cpu scratch buffer instead in order to not\n    pressure stack space any further. I do suspect that this limit was mainly\n    set in place for this particular reason. So, ease program development\n    by removing this limitation and make the scratchpad generic, so it can\n    be reused.\n\n    Signed-off-by: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Acked-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit f35e92fc2d57591123bad96b6fd88745d56639ca\nAuthor: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\nDate:   Fri Feb 19 23:05:23 2016 +0100\n\n    bpf: add generic bpf_csum_diff helper\n\n    For L4 checksums, we currently have bpf_l4_csum_replace() helper. It\u0027s\n    currently limited to handle 2 and 4 byte changes in a header and feeds the\n    from/to into inet_proto_csum_replace{2,4}() helpers of the kernel. When\n    working with IPv6, for example, this makes it rather cumbersome to deal\n    with, similarly when editing larger parts of a header.\n\n    Instead, extend the API in a more generic way: For bpf_l4_csum_replace(),\n    add a case for header field mask of 0 to change the checksum at a given\n    offset through inet_proto_csum_replace_by_diff(), and provide a helper\n    bpf_csum_diff() that can generically calculate a from/to diff for arbitrary\n    amounts of data.\n\n    This can be used in multiple ways: for the bpf_l4_csum_replace() only\n    part, this even provides us with the option to insert precalculated diffs\n    from user space f.e. from a map, or from bpf_csum_diff() during runtime.\n\n    bpf_csum_diff() has a optional from/to stack buffer input, so we can\n    calculate a diff by using a scratchbuffer for scenarios where we\u0027re\n    inserting (from is NULL), removing (to is NULL) or diffing (from/to buffers\n    don\u0027t need to be of equal size) data. Also, bpf_csum_diff() allows to\n    feed a previous csum into csum_partial(), so the function can also be\n    cascaded.\n\n    Signed-off-by: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Acked-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 9a9393327c7773ddf574cdb94c920fe246d0b169\nAuthor: Hannes Frederic Sowa \u003channes@stressinduktion.org\u003e\nDate:   Tue Apr 5 17:10:16 2016 +0200\n\n    tun: use socket locks for sk_{attach,detatch}_filter\n\n    This reverts commit 5a5abb1fa3b05dd (\"tun, bpf: fix suspicious RCU usage\n    in tun_{attach, detach}_filter\") and replaces it to use lock_sock around\n    sk_{attach,detach}_filter. The checks inside filter.c are updated with\n    lockdep_sock_is_held to check for proper socket locks.\n\n    It keeps the code cleaner by ensuring that only one lock governs the\n    socket filter instead of two independent locks.\n\n    Cc: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Signed-off-by: Hannes Frederic Sowa \u003channes@stressinduktion.org\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 5543a384f2aa671e9aeb83de915cb685517db26c\nAuthor: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\nDate:   Wed Apr 13 00:10:51 2016 +0200\n\n    bpf, verifier: add ARG_PTR_TO_RAW_STACK type\n\n    When passing buffers from eBPF stack space into a helper function, we have\n    ARG_PTR_TO_STACK argument type for helpers available. The verifier makes sure\n    that such buffers are initialized, within boundaries, etc.\n\n    However, the downside with this is that we have a couple of helper functions\n    such as bpf_skb_load_bytes() that fill out the passed buffer in the expected\n    success case anyway, so zero initializing them prior to the helper call is\n    unneeded/wasted instructions in the eBPF program that can be avoided.\n\n    Therefore, add a new helper function argument type called ARG_PTR_TO_RAW_STACK.\n    The idea is to skip the STACK_MISC check in check_stack_boundary() and color\n    the related stack slots as STACK_MISC after we checked all call arguments.\n\n    Helper functions using ARG_PTR_TO_RAW_STACK must make sure that every path of\n    the helper function will fill the provided buffer area, so that we cannot leak\n    any uninitialized stack memory. This f.e. means that error paths need to\n    memset() the buffers, but the expected fast-path doesn\u0027t have to do this\n    anymore.\n\n    Since there\u0027s no such helper needing more than at most one ARG_PTR_TO_RAW_STACK\n    argument, we can keep it simple and don\u0027t need to check for multiple areas.\n    Should in future such a use-case really appear, we have check_raw_mode() that\n    will make sure we implement support for it first.\n\n    Signed-off-by: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Acked-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 8683d7e6de79b0099bb98c7f5ea9590b2f49b200\nAuthor: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\nDate:   Wed Apr 13 00:10:50 2016 +0200\n\n    bpf, verifier: add bpf_call_arg_meta for passing meta data\n\n    Currently, when the verifier checks calls in check_call() function, we\n    call check_func_arg() for all 5 arguments e.g. to make sure expected types\n    are correct. In some cases, we collect meta data (here: map pointer) to\n    perform additional checks such as checking stack boundary on key/value\n    sizes for subsequent arguments. As we\u0027re going to extend the meta data,\n    add a generic struct bpf_call_arg_meta that we can use for passing into\n    check_func_arg().\n\n    Signed-off-by: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Acked-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 22249ef3025ba459847346d1b2bae3012caeee25\nAuthor: Alexei Starovoitov \u003cast@fb.com\u003e\nDate:   Tue Apr 12 10:26:19 2016 -0700\n\n    bpf/verifier: reject invalid LD_ABS | BPF_DW instruction\n\n    verifier must check for reserved size bits in instruction opcode and\n    reject BPF_LD | BPF_ABS | BPF_DW and BPF_LD | BPF_IND | BPF_DW instructions,\n    otherwise interpreter will WARN_RATELIMIT on them during execution.\n\n    Fixes: ddd872bc3098 (\"bpf: verifier: add checks for BPF_ABS | BPF_IND instructions\")\n    Signed-off-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Acked-by: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit ffecfe4243f1874eaa403286799be4a705c5c063\nAuthor: Alexei Starovoitov \u003cast@fb.com\u003e\nDate:   Wed Apr 6 19:39:21 2016 -0700\n\n    bpf: simplify verifier register state assignments\n\n    verifier is using the following structure to track the state of registers:\n    struct reg_state {\n        enum bpf_reg_type type;\n        union {\n            int imm;\n            struct bpf_map *map_ptr;\n        };\n    };\n    and later on in states_equal() does memcmp(\u0026old-\u003eregs[i], \u0026cur-\u003eregs[i],..)\n    to find equivalent states.\n    Throughout the code of verifier there are assignements to \u0027imm\u0027 and \u0027map_ptr\u0027\n    fields and it\u0027s not obvious that most of the assignments into \u0027imm\u0027 don\u0027t\n    need to clear extra 4 bytes (like mark_reg_unknown_value() does) to make sure\n    that memcmp doesn\u0027t go over junk left from \u0027map_ptr\u0027 assignment.\n\n    Simplify the code by converting \u0027int\u0027 into \u0027long\u0027\n\n    Suggested-by: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Signed-off-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Acked-by: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 37d179bc68b8cf62833ee391ac4da9d363a76845\nAuthor: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\nDate:   Tue Apr 5 22:33:17 2016 +0200\n\n    bpf, verifier: further improve search pruning\n\n    The verifier needs to go through every path of the program in\n    order to check that it terminates safely, which can be quite a\n    lot of instructions that need to be processed f.e. in cases with\n    more branchy programs. With search pruning from f1bca824dabb (\"bpf:\n    add search pruning optimization to verifier\") the search space can\n    already be reduced significantly when the verifier detects that\n    a previously walked path with same register and stack contents\n    terminated already (see verifier\u0027s states_equal()), so the search\n    can skip walking those states.\n\n    When working with larger programs of \u003e ~2000 (out of max 4096)\n    insns, we found that the current limit of 32k instructions is easily\n    hit. For example, a case we ran into is that the search space cannot\n    be pruned due to branches at the beginning of the program that make\n    use of certain stack space slots (STACK_MISC), which are never used\n    in the remaining program (STACK_INVALID). Therefore, the verifier\n    needs to walk paths for the slots in STACK_INVALID state, but also\n    all remaining paths with a stack structure, where the slots are in\n    STACK_MISC, which can nearly double the search space needed. After\n    various experiments, we find that a limit of 64k processed insns is\n    a more reasonable choice when dealing with larger programs in practice.\n    This still allows to reject extreme crafted cases that can have a\n    much higher complexity (f.e. \u003e ~300k) within the 4096 insns limit\n    due to search pruning not being able to take effect.\n\n    Furthermore, we found that a lot of states can be pruned after a\n    call instruction, f.e. we were able to reduce the search state by\n    ~35% in some cases with this heuristic, trade-off is to keep a bit\n    more states in env-\u003eexplored_states. Usually, call instructions\n    have a number of preceding register assignments and/or stack stores,\n    where search pruning has a better chance to suceed in states_equal()\n    test. The current code marks the branch targets with STATE_LIST_MARK\n    in case of conditional jumps, and the next (t + 1) instruction in\n    case of unconditional jump so that f.e. a backjump will walk it. We\n    also did experiments with using t + insns[t].off + 1 as a marker in\n    the unconditionally jump case instead of t + 1 with the rationale\n    that these two branches of execution that converge after the label\n    might have more potential of pruning. We found that it was a bit\n    better, but not necessarily significantly better than the current\n    state, perhaps also due to clang not generating back jumps often.\n    Hence, we left that as is for now.\n\n    Signed-off-by: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Acked-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 94a64d7f40731dc185f583fbe4f2a09568ae320c\nAuthor: Alexei Starovoitov \u003cast@fb.com\u003e\nDate:   Wed Apr 6 18:43:28 2016 -0700\n\n    bpf: sanitize bpf tracepoint access\n\n    during bpf program loading remember the last byte of ctx access\n    and at the time of attaching the program to tracepoint check that\n    the program doesn\u0027t access bytes beyond defined in tracepoint fields\n\n    This also disallows access to __dynamic_array fields, but can be\n    relaxed in the future.\n\n    Signed-off-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 69530735cc0048a79e0ea28f2f5ab3d0b6ad3580\nAuthor: Alexei Starovoitov \u003cast@fb.com\u003e\nDate:   Wed Apr 6 18:43:27 2016 -0700\n\n    bpf: support bpf_get_stackid() and bpf_perf_event_output() in tracepoint programs\n\n    needs two wrapper functions to fetch \u0027struct pt_regs *\u0027 to convert\n    tracepoint bpf context into kprobe bpf context to reuse existing\n    helper functions\n\n    Signed-off-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit db2a5763f32413dbc9148c5c4dc5ffbc0b967d78\nAuthor: Alexei Starovoitov \u003cast@fb.com\u003e\nDate:   Wed Apr 6 18:43:26 2016 -0700\n\n    bpf: register BPF_PROG_TYPE_TRACEPOINT program type\n\n    register tracepoint bpf program type and let it call the same set\n    of helper functions as BPF_PROG_TYPE_KPROBE\n\n    Signed-off-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 03365d848f0a3753b001bbb72c1e45b0f83534f8\nAuthor: Alexei Starovoitov \u003cast@fb.com\u003e\nDate:   Wed Apr 6 18:43:25 2016 -0700\n\n    perf, bpf: allow bpf programs attach to tracepoints\n\n    introduce BPF_PROG_TYPE_TRACEPOINT program type and allow it to be attached\n    to the perf tracepoint handler, which will copy the arguments into\n    the per-cpu buffer and pass it to the bpf program as its first argument.\n    The layout of the fields can be discovered by doing\n    \u0027cat /sys/kernel/debug/tracing/events/sched/sched_switch/format\u0027\n    prior to the compilation of the program with exception that first 8 bytes\n    are reserved and not accessible to the program. This area is used to store\n    the pointer to \u0027struct pt_regs\u0027 which some of the bpf helpers will use:\n    +---------+\n    | 8 bytes | hidden \u0027struct pt_regs *\u0027 (inaccessible to bpf program)\n    +---------+\n    | N bytes | static tracepoint fields defined in tracepoint/format (bpf readonly)\n    +---------+\n    | dynamic | __dynamic_array bytes of tracepoint (inaccessible to bpf yet)\n    +---------+\n\n    Not that all of the fields are already dumped to user space via perf ring buffer\n    and broken application access it directly without consulting tracepoint/format.\n    Same rule applies here: static tracepoint fields should only be accessed\n    in a format defined in tracepoint/format. The order of fields and\n    field sizes are not an ABI.\n\n    Signed-off-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Acked-by: Peter Zijlstra (Intel) \u003cpeterz@infradead.org\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 986f3aa294b51f80059c67b9bac4c80cbc0789eb\nAuthor: Alexei Starovoitov \u003cast@fb.com\u003e\nDate:   Wed Apr 6 18:43:24 2016 -0700\n\n    perf: split perf_trace_buf_prepare into alloc and update parts\n\n    split allows to move expensive update of \u0027struct trace_entry\u0027 to later phase.\n    Repurpose unused 1st argument of perf_tp_event() to indicate event type.\n\n    While splitting use temp variable \u0027rctx\u0027 instead of \u0027*rctx\u0027 to avoid\n    unnecessary loads done by the compiler due to -fno-strict-aliasing\n\n    Signed-off-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Acked-by: Peter Zijlstra (Intel) \u003cpeterz@infradead.org\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 1870873f44e69836a5677f302b85f0cc830b4a84\nAuthor: Alexei Starovoitov \u003cast@fb.com\u003e\nDate:   Wed Apr 6 18:43:23 2016 -0700\n\n    perf: remove unused __addr variable\n\n    now all calls to perf_trace_buf_submit() pass 0 as 4th\n    argument which will be repurposed in the next patch which will\n    change the meaning of 1st arg of perf_tp_event() to event_type\n\n    Signed-off-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Acked-by: Peter Zijlstra (Intel) \u003cpeterz@infradead.org\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit e7f515b54cb3de36385de8b42c2690531931be2f\nAuthor: Al Viro \u003cviro@zeniv.linux.org.uk\u003e\nDate:   Fri Mar 25 12:06:51 2016 -0400\n\n    bpf: reject invalid names right in -\u003elookup()\n\n    ... and other methods won\u0027t see them at all\n\n    Signed-off-by: Al Viro \u003cviro@zeniv.linux.org.uk\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 989de5ab3bd4b747a39a979e052203e3980593c4\nAuthor: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\nDate:   Fri Mar 25 00:30:25 2016 +0100\n\n    bpf: add missing map_flags to bpf_map_show_fdinfo\n\n    Add map_flags attribute to bpf_map_show_fdinfo(), so that tools like\n    tc can check for them when loading objects from a pinned entry, e.g.\n    if user intent wrt allocation (BPF_F_NO_PREALLOC) is different to the\n    pinned object, it can bail out. Follow-up to 6c9059817432 (\"bpf:\n    pre-allocate hash map elements\"), so that tc can still support this\n    with v4.6.\n\n    Signed-off-by: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Acked-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 6b10fb1009db25dd0d3c5dd7ecfd422591599413\nAuthor: Alexei Starovoitov \u003cast@fb.com\u003e\nDate:   Wed Mar 9 20:02:33 2016 -0800\n\n    bpf: avoid copying junk bytes in bpf_get_current_comm()\n\n    Lots of places in the kernel use memcpy(buf, comm, TASK_COMM_LEN); but\n    the result is typically passed to print(\"%s\", buf) and extra bytes\n    after zero don\u0027t cause any harm.\n    In bpf the result of bpf_get_current_comm() is used as the part of\n    map key and was causing spurious hash map mismatches.\n    Use strlcpy() to guarantee zero-terminated string.\n    bpf verifier checks that output buffer is zero-initialized,\n    so even for short task names the output buffer don\u0027t have junk bytes.\n    Note it\u0027s not a security concern, since kprobe+bpf is root only.\n\n    Fixes: ffeedafbf023 (\"bpf: introduce current-\u003epid, tgid, uid, gid, comm accessors\")\n    Reported-by: Tobias Waldekranz \u003ctobias@waldekranz.com\u003e\n    Signed-off-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit b1d5a61adf9e790e2d255158bb19706367de9329\nAuthor: Alexei Starovoitov \u003cast@fb.com\u003e\nDate:   Wed Mar 9 18:56:49 2016 -0800\n\n    bpf: bpf_stackmap_copy depends on CONFIG_PERF_EVENTS\n\n    0-day bot reported build error:\n    kernel/built-in.o: In function `map_lookup_elem\u0027:\n    \u003e\u003e kernel/bpf/.tmp_syscall.o:(.text+0x329b3c): undefined reference to `bpf_stackmap_copy\u0027\n    when CONFIG_BPF_SYSCALL is set and CONFIG_PERF_EVENTS is not.\n    Add weak definition to resolve it.\n    This code path in map_lookup_elem() is never taken\n    when CONFIG_PERF_EVENTS is not set.\n\n    Fixes: 557c0c6e7df8 (\"bpf: convert stackmap to pre-allocation\")\n    Reported-by: Fengguang Wu \u003cfengguang.wu@intel.com\u003e\n    Signed-off-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit d7f8276b4034c326e5770c41b4b81461b02e6e22\nAuthor: Alexei Starovoitov \u003cast@fb.com\u003e\nDate:   Mon Mar 7 21:57:17 2016 -0800\n\n    bpf: convert stackmap to pre-allocation\n\n    It was observed that calling bpf_get_stackid() from a kprobe inside\n    slub or from spin_unlock causes similar deadlock as with hashmap,\n    therefore convert stackmap to use pre-allocated memory.\n\n    The call_rcu is no longer feasible mechanism, since delayed freeing\n    causes bpf_get_stackid() to fail unpredictably when number of actual\n    stacks is significantly less than user requested max_entries.\n    Since elements are no longer freed into slub, we can push elements into\n    freelist immediately and let them be recycled.\n    However the very unlikley race between user space map_lookup() and\n    program-side recycling is possible:\n         cpu0                          cpu1\n         ----                          ----\n    user does lookup(stackidX)\n    starts copying ips into buffer\n                                       delete(stackidX)\n                                       calls bpf_get_stackid()\n    \t\t\t\t   which recyles the element and\n                                       overwrites with new stack trace\n\n    To avoid user space seeing a partial stack trace consisting of two\n    merged stack traces, do bucket \u003d xchg(, NULL); copy; xchg(,bucket);\n    to preserve consistent stack trace delivery to user space.\n    Now we can move memset(,0) of left-over element value from critical\n    path of bpf_get_stackid() into slow-path of user space lookup.\n    Also disallow lookup() from bpf program, since it\u0027s useless and\n    program shouldn\u0027t be messing with collected stack trace.\n\n    Note that similar race between user space lookup and kernel side updates\n    is also present in hashmap, but it\u0027s not a new race. bpf programs were\n    always allowed to modify hash and array map elements while user space\n    is copying them.\n\n    Fixes: d5a3b1f69186 (\"bpf: introduce BPF_MAP_TYPE_STACK_TRACE\")\n    Signed-off-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit c8e47381bd3af50fd4cde463003f44e9cf278a9b\nAuthor: Alexei Starovoitov \u003cast@fb.com\u003e\nDate:   Mon Mar 7 21:57:16 2016 -0800\n\n    bpf: check for reserved flag bits in array and stack maps\n\n    Suggested-by: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Signed-off-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit b5e87bad280370ef1096df8461836a634346a2c2\nAuthor: Alexei Starovoitov \u003cast@fb.com\u003e\nDate:   Mon Mar 7 21:57:15 2016 -0800\n\n    bpf: pre-allocate hash map elements\n\n    If kprobe is placed on spin_unlock then calling kmalloc/kfree from\n    bpf programs is not safe, since the following dead lock is possible:\n    kfree-\u003espin_lock(kmem_cache_node-\u003elock)...spin_unlock-\u003ekprobe-\u003e\n    bpf_prog-\u003emap_update-\u003ekmalloc-\u003espin_lock(of the same kmem_cache_node-\u003elock)\n    and deadlocks.\n\n    The following solutions were considered and some implemented, but\n    eventually discarded\n    - kmem_cache_create for every map\n    - add recursion check to slow-path of slub\n    - use reserved memory in bpf_map_update for in_irq or in preempt_disabled\n    - kmalloc via irq_work\n\n    At the end pre-allocation of all map elements turned out to be the simplest\n    solution and since the user is charged upfront for all the memory, such\n    pre-allocation doesn\u0027t affect the user space visible behavior.\n\n    Since it\u0027s impossible to tell whether kprobe is triggered in a safe\n    location from kmalloc point of view, use pre-allocation by default\n    and introduce new BPF_F_NO_PREALLOC flag.\n\n    While testing of per-cpu hash maps it was discovered\n    that alloc_percpu(GFP_ATOMIC) has odd corner cases and often\n    fails to allocate memory even when 90% of it is free.\n    The pre-allocation of per-cpu hash elements solves this problem as well.\n\n    Turned out that bpf_map_update() quickly followed by\n    bpf_map_lookup()+bpf_map_delete() is very common pattern used\n    in many of iovisor/bcc/tools, so there is additional benefit of\n    pre-allocation, since such use cases are must faster.\n\n    Since all hash map elements are now pre-allocated we can remove\n    atomic increment of htab-\u003ecount and save few more cycles.\n\n    Also add bpf_map_precharge_memlock() to check rlimit_memlock early to avoid\n    large malloc/free done by users who don\u0027t have sufficient limits.\n\n    Pre-allocation is done with vmalloc and alloc/free is done\n    via percpu_freelist. Here are performance numbers for different\n    pre-allocation algorithms that were implemented, but discarded\n    in favor of percpu_freelist:\n\n    1 cpu:\n    pcpu_ida\t2.1M\n    pcpu_ida nolock\t2.3M\n    bt\t\t2.4M\n    kmalloc\t\t1.8M\n    hlist+spinlock\t2.3M\n    pcpu_freelist\t2.6M\n\n    4 cpu:\n    pcpu_ida\t1.5M\n    pcpu_ida nolock\t1.8M\n    bt w/smp_align\t1.7M\n    bt no/smp_align\t1.1M\n    kmalloc\t\t0.7M\n    hlist+spinlock\t0.2M\n    pcpu_freelist\t2.0M\n\n    8 cpu:\n    pcpu_ida\t0.7M\n    bt w/smp_align\t0.8M\n    kmalloc\t\t0.4M\n    pcpu_freelist\t1.5M\n\n    32 cpu:\n    kmalloc\t\t0.13M\n    pcpu_freelist\t0.49M\n\n    pcpu_ida nolock is a modified percpu_ida algorithm without\n    percpu_ida_cpu locks and without cross-cpu tag stealing.\n    It\u0027s faster than existing percpu_ida, but not as fast as pcpu_freelist.\n\n    bt is a variant of block/blk-mq-tag.c simlified and customized\n    for bpf use case. bt w/smp_align is using cache line for every \u0027long\u0027\n    (similar to blk-mq-tag). bt no/smp_align allocates \u0027long\u0027\n    bitmasks continuously to save memory. It\u0027s comparable to percpu_ida\n    and in some cases faster, but slower than percpu_freelist\n\n    hlist+spinlock is the simplest free list with single spinlock.\n    As expeceted it has very bad scaling in SMP.\n\n    kmalloc is existing implementation which is still available via\n    BPF_F_NO_PREALLOC flag. It\u0027s significantly slower in single cpu and\n    in 8 cpu setup it\u0027s 3 times slower than pre-allocation with pcpu_freelist,\n    but saves memory, so in cases where map-\u003emax_entries can be large\n    and number of map update/delete per second is low, it may make\n    sense to use it.\n\n    Signed-off-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 8ab0075ed6148a1c03aa0cbdb46d87d80dffcffc\nAuthor: Alexei Starovoitov \u003cast@fb.com\u003e\nDate:   Mon Mar 7 21:57:14 2016 -0800\n\n    bpf: introduce percpu_freelist\n\n    Introduce simple percpu_freelist to keep single list of elements\n    spread across per-cpu singly linked lists.\n\n    /* push element into the list */\n    void pcpu_freelist_push(struct pcpu_freelist *, struct pcpu_freelist_node *);\n\n    /* pop element from the list */\n    struct pcpu_freelist_node *pcpu_freelist_pop(struct pcpu_freelist *);\n\n    The object is pushed to the current cpu list.\n    Pop first trying to get the object from the current cpu list,\n    if it\u0027s empty goes to the neigbour cpu list.\n\n    For bpf program usage pattern the collision rate is very low,\n    since programs push and pop the objects typically on the same cpu.\n\n    Signed-off-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit a003f77a9e901023841778cc4388294fb537d650\nAuthor: Alexei Starovoitov \u003cast@fb.com\u003e\nDate:   Mon Mar 7 21:57:13 2016 -0800\n\n    bpf: prevent kprobe+bpf deadlocks\n\n    if kprobe is placed within update or delete hash map helpers\n    that hold bucket spin lock and triggered bpf program is trying to\n    grab the spinlock for the same bucket on the same cpu, it will\n    deadlock.\n    Fix it by extending existing recursion prevention mechanism.\n\n    Note, map_lookup and other tracing helpers don\u0027t have this problem,\n    since they don\u0027t hold any locks and don\u0027t modify global data.\n    bpf_trace_printk has its own recursive check and ok as well.\n\n    Signed-off-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Acked-by: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit d9cd772dca5eb3a4e03fa868aa2d6add480d817a\nAuthor: Josh Poimboeuf \u003cjpoimboe@redhat.com\u003e\nDate:   Sun Feb 28 22:22:37 2016 -0600\n\n    bpf: Mark __bpf_prog_run() stack frame as non-standard\n\n    objtool reports the following false positive warnings:\n\n      kernel/bpf/core.o: warning: objtool: __bpf_prog_run()+0x5c: sibling call from callable instruction with changed frame pointer\n      kernel/bpf/core.o: warning: objtool: __bpf_prog_run()+0x60: function has unreachable instruction\n      kernel/bpf/core.o: warning: objtool: __bpf_prog_run()+0x64: function has unreachable instruction\n      [...]\n\n    It\u0027s confused by the following dynamic jump instruction in\n    __bpf_prog_run()::\n\n      jmp     *(%r12,%rax,8)\n\n    which corresponds to the following line in the C code:\n\n      goto *jumptable[insn-\u003ecode];\n\n    There\u0027s no way for objtool to deterministically find all possible\n    branch targets for a dynamic jump, so it can\u0027t verify this code.\n\n    In this case the jumps all stay within the function, and there\u0027s nothing\n    unusual going on related to the stack, so we can whitelist the function.\n\n    Signed-off-by: Josh Poimboeuf \u003cjpoimboe@redhat.com\u003e\n    Acked-by: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Acked-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Cc: Andrew Morton \u003cakpm@linux-foundation.org\u003e\n    Cc: Andy Lutomirski \u003cluto@kernel.org\u003e\n    Cc: Arnaldo Carvalho de Melo \u003cacme@kernel.org\u003e\n    Cc: Bernd Petrovitsch \u003cbernd@petrovitsch.priv.at\u003e\n    Cc: Borislav Petkov \u003cbp@alien8.de\u003e\n    Cc: Chris J Arges \u003cchris.j.arges@canonical.com\u003e\n    Cc: Jiri Slaby \u003cjslaby@suse.cz\u003e\n    Cc: Linus Torvalds \u003ctorvalds@linux-foundation.org\u003e\n    Cc: Michal Marek \u003cmmarek@suse.cz\u003e\n    Cc: Namhyung Kim \u003cnamhyung@gmail.com\u003e\n    Cc: Pedro Alves \u003cpalves@redhat.com\u003e\n    Cc: Peter Zijlstra \u003cpeterz@infradead.org\u003e\n    Cc: Thomas Gleixner \u003ctglx@linutronix.de\u003e\n    Cc: live-patching@vger.kernel.org\n    Cc: netdev@vger.kernel.org\n    Link: http://lkml.kernel.org/r/b90e6bf3fdbfb5c4cc1b164b965502e53cf48935.1456719558.git.jpoimboe@redhat.com\n    Signed-off-by: Ingo Molnar \u003cmingo@kernel.org\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 33df8c4ab343871610c65197264e8cc09ef04612\nAuthor: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\nDate:   Fri Feb 19 23:05:22 2016 +0100\n\n    bpf: add new arg_type that allows for 0 sized stack buffer\n\n    Currently, when we pass a buffer from the eBPF stack into a helper\n    function, the function proto indicates argument types as ARG_PTR_TO_STACK\n    and ARG_CONST_STACK_SIZE pair. If R\u003cX\u003e contains the former, then R\u003cX+1\u003e\n    must be of the latter type. Then, verifier checks whether the buffer\n    points into eBPF stack, is initialized, etc. The verifier also guarantees\n    that the constant value passed in R\u003cX+1\u003e is greater than 0, so helper\n    functions don\u0027t need to test for it and can always assume a non-NULL\n    initialized buffer as well as non-0 buffer size.\n\n    This patch adds a new argument types ARG_CONST_STACK_SIZE_OR_ZERO that\n    allows to also pass NULL as R\u003cX\u003e and 0 as R\u003cX+1\u003e into the helper function.\n    Such helper functions, of course, need to be able to handle these cases\n    internally then. Verifier guarantees that either R\u003cX\u003e \u003d\u003d NULL \u0026\u0026 R\u003cX+1\u003e \u003d\u003d 0\n    or R\u003cX\u003e !\u003d NULL \u0026\u0026 R\u003cX+1\u003e !\u003d 0 (like the case of ARG_CONST_STACK_SIZE), any\n    other combinations are not possible to load.\n\n    I went through various options of extending the verifier, and introducing\n    the type ARG_CONST_STACK_SIZE_OR_ZERO seems to have most minimal changes\n    needed to the verifier.\n\n    Signed-off-by: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Acked-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 43e79743a6b6d486b3977fd0c281be531096ab5a\nAuthor: Alexei Starovoitov \u003cast@fb.com\u003e\nDate:   Wed Feb 17 19:58:58 2016 -0800\n\n    bpf: introduce BPF_MAP_TYPE_STACK_TRACE\n\n    add new map type to store stack traces and corresponding helper\n    bpf_get_stackid(ctx, map, flags) - walk user or kernel stack and return id\n    @ctx: struct pt_regs*\n    @map: pointer to stack_trace map\n    @flags: bits 0-7 - numer of stack frames to skip\n            bit 8 - collect user stack instead of kernel\n            bit 9 - compare stacks by hash only\n            bit 10 - if two different stacks hash into the same stackid\n                     discard old\n            other bits - reserved\n    Return: \u003e\u003d 0 stackid on success or negative error\n\n    stackid is a 32-bit integer handle that can be further combined with\n    other data (including other stackid) and used as a key into maps.\n\n    Userspace will access stackmap using standard lookup/delete syscall commands to\n    retrieve full stack trace for given stackid.\n\n    Signed-off-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 4c16860477822114cf19d85c45c61db73e5d88e3\nAuthor: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\nDate:   Mon Jan 11 01:16:39 2016 +0100\n\n    bpf: support ipv6 for bpf_skb_{set,get}_tunnel_key\n\n    After IPv6 support has recently been added to metadata dst and related\n    encaps, add support for populating/reading it from an eBPF program.\n\n    Commit d3aa45ce6b (\"bpf: add helpers to access tunnel metadata\") started\n    with initial IPv4-only support back then (due to IPv6 metadata support\n    not being available yet).\n\n    To stay compatible with older programs, we need to test for the passed\n    structure size. Also TOS and TTL support from the ip_tunnel_info key has\n    been added. Tested with vxlan devs in collect meta data mode with IPv4,\n    IPv6 and in compat mode over different network namespaces.\n\n    Signed-off-by: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Acked-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 3e5d2da2428d9a2ebc15119db5de5be9b59f857e\nAuthor: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\nDate:   Mon Jan 11 01:16:38 2016 +0100\n\n    bpf: export helper function flags and reject invalid ones\n\n    Export flags used by eBPF helper functions through UAPI, so they can be\n    used by programs (instead of them redefining all flags each time or just\n    using the hard-coded values). It also gives a better overview what flags\n    are used where and we can further get rid of the extra macros defined in\n    filter.c. Moreover, reject invalid flags.\n\n    Signed-off-by: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Acked-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 702bf27a7cf50f7496f476041e6c404f9ebf49c5\nAuthor: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\nDate:   Thu Jan 7 15:50:23 2016 +0100\n\n    bpf: add skb_postpush_rcsum and fix dev_forward_skb occasions\n\n    Add a small helper skb_postpush_rcsum() and fix up redirect locations\n    that need CHECKSUM_COMPLETE fixups on ingress. dev_forward_skb() expects\n    a proper csum that covers also Ethernet header, f.e. since 2c26d34bbcc0\n    (\"net/core: Handle csum for CHECKSUM_COMPLETE VXLAN forwarding\"), we\n    also do skb_postpull_rcsum() after pulling Ethernet header off via\n    eth_type_trans().\n\n    When using eBPF in a netns setup f.e. with vxlan in collect metadata mode,\n    I can trigger the following csum issue with an IPv6 setup:\n\n      [  505.144065] dummy1: hw csum failure\n      [...]\n      [  505.144108] Call Trace:\n      [  505.144112]  \u003cIRQ\u003e  [\u003cffffffff81372f08\u003e] dump_stack+0x44/0x5c\n      [  505.144134]  [\u003cffffffff81607cea\u003e] netdev_rx_csum_fault+0x3a/0x40\n      [  505.144142]  [\u003cffffffff815fee3f\u003e] __skb_checksum_complete+0xcf/0xe0\n      [  505.144149]  [\u003cffffffff816f0902\u003e] nf_ip6_checksum+0xb2/0x120\n      [  505.144161]  [\u003cffffffffa08c0e0e\u003e] icmpv6_error+0x17e/0x328 [nf_conntrack_ipv6]\n      [  505.144170]  [\u003cffffffffa0898eca\u003e] ? ip6t_do_table+0x2fa/0x645 [ip6_tables]\n      [  505.144177]  [\u003cffffffffa08c0725\u003e] ? ipv6_get_l4proto+0x65/0xd0 [nf_conntrack_ipv6]\n      [  505.144189]  [\u003cffffffffa06c9a12\u003e] nf_conntrack_in+0xc2/0x5a0 [nf_conntrack]\n      [  505.144196]  [\u003cffffffffa08c039c\u003e] ipv6_conntrack_in+0x1c/0x20 [nf_conntrack_ipv6]\n      [  505.144204]  [\u003cffffffff8164385d\u003e] nf_iterate+0x5d/0x70\n      [  505.144210]  [\u003cffffffff816438d6\u003e] nf_hook_slow+0x66/0xc0\n      [  505.144218]  [\u003cffffffff816bd302\u003e] ipv6_rcv+0x3f2/0x4f0\n      [  505.144225]  [\u003cffffffff816bca40\u003e] ? ip6_make_skb+0x1b0/0x1b0\n      [  505.144232]  [\u003cffffffff8160b77b\u003e] __netif_receive_skb_core+0x36b/0x9a0\n      [  505.144239]  [\u003cffffffff8160bdc8\u003e] ? __netif_receive_skb+0x18/0x60\n      [  505.144245]  [\u003cffffffff8160bdc8\u003e] __netif_receive_skb+0x18/0x60\n      [  505.144252]  [\u003cffffffff8160ccff\u003e] process_backlog+0x9f/0x140\n      [  505.144259]  [\u003cffffffff8160c4a5\u003e] net_rx_action+0x145/0x320\n      [...]\n\n    What happens is that on ingress, we push Ethernet header back in, either\n    from cls_bpf or right before skb_do_redirect(), but without updating csum.\n    The \"hw csum failure\" can be fixed by using the new skb_postpush_rcsum()\n    helper for the dev_forward_skb() case to correct the csum diff again.\n\n    Thanks to Hannes Frederic Sowa for the csum_partial() idea!\n\n    Fixes: 3896d655f4d4 (\"bpf: introduce bpf_clone_redirect() helper\")\n    Fixes: 27b29f63058d (\"bpf: add bpf_redirect() helper\")\n    Signed-off-by: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Acked-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 1c1049b97190ebeb4ff711f76db6f89c6aa68b92\nAuthor: Craig Gallek \u003ckraig@google.com\u003e\nDate:   Mon Jan 4 17:41:47 2016 -0500\n\n    soreuseport: setsockopt SO_ATTACH_REUSEPORT_[CE]BPF\n\n    Expose socket options for setting a classic or extended BPF program\n    for use when selecting sockets in an SO_REUSEPORT group.  These options\n    can be used on the first socket to belong to a group before bind or\n    on any socket in the group after bind.\n\n    This change includes refactoring of the existing sk_filter code to\n    allow reuse of the existing BPF filter validation checks.\n\n    Signed-off-by: Craig Gallek \u003ckraig@google.com\u003e\n    Acked-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n    Change-Id: If4bd3258a57eeca4622a135d9cdf64ecc12f8a27\n\ncommit 3a46d6c7aeeac4486b65f682a1b6e7e97e7740a6\nAuthor: Craig Gallek \u003ckraig@google.com\u003e\nDate:   Mon Jan 4 17:41:46 2016 -0500\n\n    soreuseport: fast reuseport UDP socket selection\n\n    Include a struct sock_reuseport instance when a UDP socket binds to\n    a specific address for the first time with the reuseport flag set.\n    When selecting a socket for an incoming UDP packet, use the information\n    available in sock_reuseport if present.\n\n    This required adding an additional field to the UDP source address\n    equality function to differentiate between exact and wildcard matches.\n    The original use case allowed wildcard matches when checking for\n    existing port uses during bind.  The new use case of adding a socket\n    to a reuseport group requires exact address matching.\n\n    Performance test (using a machine with 2 CPU sockets and a total of\n    48 cores):  Create reuseport groups of varying size.  Use one socket\n    from this group per user thread (pinning each thread to a different\n    core) calling recvmmsg in a tight loop.  Record number of messages\n    received per second while saturating a 10G link.\n      10 sockets: 18% increase (~2.8M -\u003e 3.3M pkts/s)\n      20 sockets: 14% increase (~2.9M -\u003e 3.3M pkts/s)\n      40 sockets: 13% increase (~3.0M -\u003e 3.4M pkts/s)\n\n    This work is based off a similar implementation written by\n    Ying Cai \u003cycai@google.com\u003e for implementing policy-based reuseport\n    selection.\n\n    Signed-off-by: Craig Gallek \u003ckraig@google.com\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 93cf2d5634671797a50a13a428db67e34d249183\nAuthor: Craig Gallek \u003ckraig@google.com\u003e\nDate:   Mon Jan 4 17:41:45 2016 -0500\n\n    soreuseport: define reuseport groups\n\n    struct sock_reuseport is an optional shared structure referenced by each\n    socket belonging to a reuseport group.  When a socket is bound to an\n    address/port not yet in use and the reuseport flag has been set, the\n    structure will be allocated and attached to the newly bound socket.\n    When subsequent calls to bind are made for the same address/port, the\n    shared structure will be updated to include the new socket and the\n    newly bound socket will reference the group structure.\n\n    Usually, when an incoming packet was destined for a reuseport group,\n    all sockets in the same group needed to be considered before a\n    dispatching decision was made.  With this structure, an appropriate\n    socket can be found after looking up just one socket in the group.\n\n    This shared structure will also allow for more complicated decisions to\n    be made when selecting a socket (eg a BPF filter).\n\n    This work is based off a similar implementation written by\n    Ying Cai \u003cycai@google.com\u003e for implementing policy-based reuseport\n    selection.\n\n    Signed-off-by: Craig Gallek \u003ckraig@google.com\u003e\n    Acked-by: Eric Dumazet \u003cedumazet@google.com\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit e85ab7e8c3669f6f35a5d056aa9157485d7e17d0\nAuthor: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\nDate:   Thu Dec 17 23:51:55 2015 +0100\n\n    bpf: fix misleading comment in bpf_convert_filter\n\n    Comment says \"User BPF\u0027s register A is mapped to our BPF register 6\",\n    which is actually wrong as the mapping is on register 0. This can\n    already be inferred from the code itself. So just remove it before\n    someone makes assumptions based on that. Only code tells truth. ;)\n\n    Signed-off-by: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Acked-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 081e8e95f12d42e1c37c20d9efd254fb48a56f4b\nAuthor: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\nDate:   Thu Dec 17 23:51:53 2015 +0100\n\n    bpf: add bpf_skb_load_bytes helper\n\n    When hacking tc programs with eBPF, one of the issues that come up\n    from time to time is to load addresses from headers. In eBPF as in\n    classic BPF, we have BPF_LD | BPF_ABS | BPF_{B,H,W} instructions that\n    extract a byte, half-word or word out of the skb data though helpers\n    such as bpf_load_pointer() (interpreter case).\n\n    F.e. extracting a whole IPv6 address could possibly look like ...\n\n      union v6addr {\n        struct {\n          __u32 p1;\n          __u32 p2;\n          __u32 p3;\n          __u32 p4;\n        };\n        __u8 addr[16];\n      };\n\n      [...]\n\n      a.p1 \u003d htonl(load_word(skb, off));\n      a.p2 \u003d htonl(load_word(skb, off +  4));\n      a.p3 \u003d htonl(load_word(skb, off +  8));\n      a.p4 \u003d htonl(load_word(skb, off + 12));\n\n      [...]\n\n      /* access to a.addr[...] */\n\n    This work adds a complementary helper bpf_skb_load_bytes() (we also\n    have bpf_skb_store_bytes()) as an alternative where the same call\n    would look like from an eBPF program:\n\n      ret \u003d bpf_skb_load_bytes(skb, off, addr, sizeof(addr));\n\n    Same verifier restrictions apply as in ffeedafbf023 (\"bpf: introduce\n    current-\u003epid, tgid, uid, gid, comm accessors\") case, where stack memory\n    access needs to be statically verified and thus guaranteed to be\n    initialized in first use (otherwise verifier cannot tell whether a\n    subsequent access to it is valid or not as it\u0027s runtime dependent).\n\n    Signed-off-by: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Acked-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit fa6dc8f675002d5d5d1d77be5fc430542b972bdb\nAuthor: Sasha Levin \u003csasha.levin@oracle.com\u003e\nDate:   Fri Feb 19 13:53:10 2016 -0500\n\n    bpf: grab rcu read lock for bpf_percpu_hash_update\n\n    bpf_percpu_hash_update() expects rcu lock to be held and warns if it\u0027s not,\n    which pointed out a missing rcu read lock.\n\n    Fixes: 15a07b338 (\"bpf: add lookup/update support for per-cpu hash and array maps\")\n    Signed-off-by: Sasha Levin \u003csasha.levin@oracle.com\u003e\n    Acked-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Acked-by: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 3aa373d0a667e624f95159af7fd40e8920cb2e92\nAuthor: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\nDate:   Wed Feb 10 16:47:11 2016 +0100\n\n    bpf: fix branch offset adjustment on backjumps after patching ctx expansion\n\n    When ctx access is used, the kernel often needs to expand/rewrite\n    instructions, so after that patching, branch offsets have to be\n    adjusted for both forward and backward jumps in the new eBPF program,\n    but for backward jumps it fails to account the delta. Meaning, for\n    example, if the expansion happens exactly on the insn that sits at\n    the jump target, it doesn\u0027t fix up the back jump offset.\n\n    Analysis on what the check in adjust_branches() is currently doing:\n\n      /* adjust offset of jmps if necessary */\n      if (i \u003c pos \u0026\u0026 i + insn-\u003eoff + 1 \u003e pos)\n        insn-\u003eoff +\u003d delta;\n      else if (i \u003e pos \u0026\u0026 i + insn-\u003eoff + 1 \u003c pos)\n        insn-\u003eoff -\u003d delta;\n\n    First condition (forward jumps):\n\n      Before:                         After:\n\n      insns[0]                        insns[0]\n      insns[1] \u003c--- i/insn            insns[1] \u003c--- i/insn\n      insns[2] \u003c--- pos               insns[P] \u003c--- pos\n      insns[3]                        insns[P]  `------| delta\n      insns[4] \u003c--- target_X          insns[P]   `-----|\n      insns[5]                        insns[3]\n                                      insns[4] \u003c--- target_X\n                                      insns[5]\n\n    First case is if we cross pos-boundary and the jump instruction was\n    before pos. This is handeled correctly. I.e. if i \u003d\u003d pos, then this\n    would mean our jump that we currently check was the patchlet itself\n    that we just injected. Since such patchlets are self-contained and\n    have no awareness of any insns before or after the patched one, the\n    delta is correctly not adjusted. Also, for the second condition in\n    case of i + insn-\u003eoff + 1 \u003d\u003d pos, means we jump to that newly patched\n    instruction, so no offset adjustment are needed. That part is correct.\n\n    Second condition (backward jumps):\n\n      Before:                         After:\n\n      insns[0]                        insns[0]\n      insns[1] \u003c--- target_X          insns[1] \u003c--- target_X\n      insns[2] \u003c--- pos \u003c-- target_Y  insns[P] \u003c--- pos \u003c-- target_Y\n      insns[3]                        insns[P]  `------| delta\n      insns[4] \u003c--- i/insn            insns[P]   `-----|\n      insns[5]                        insns[3]\n                                      insns[4] \u003c--- i/insn\n                                      insns[5]\n\n    Second interesting case is where we cross pos-boundary and the jump\n    instruction was after pos. Backward jump with i \u003d\u003d pos would be\n    impossible and pose a bug somewhere in the patchlet, so the first\n    condition checking i \u003e pos is okay only by itself. However, i +\n    insn-\u003eoff + 1 \u003c pos does not always work as intended to trigger the\n    adjustment. It works when jump targets would be far off where the\n    delta wouldn\u0027t matter. But, for example, where the fixed insn-\u003eoff\n    before pointed to pos (target_Y), it now points to pos + delta, so\n    that additional room needs to be taken into account for the check.\n    This means that i) both tests here need to be adjusted into pos + delta,\n    and ii) for the second condition, the test needs to be \u003c\u003d as pos\n    itself can be a target in the backjump, too.\n\n    Fixes: 9bac3d6d548e (\"bpf: allow extended BPF programs access skb fields\")\n    Signed-off-by: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit ca132c344ca9dfdddf927c4cc9cd1790f1849000\nAuthor: Alexei Starovoitov \u003cast@fb.com\u003e\nDate:   Mon Feb 1 22:39:55 2016 -0800\n\n    bpf: add lookup/update support for per-cpu hash and array maps\n\n    The functions bpf_map_lookup_elem(map, key, value) and\n    bpf_map_update_elem(map, key, value, flags) need to get/set\n    values from all-cpus for per-cpu hash and array maps,\n    so that user space can aggregate/update them as necessary.\n\n    Example of single counter aggregation in user space:\n      unsigned int nr_cpus \u003d sysconf(_SC_NPROCESSORS_CONF);\n      long values[nr_cpus];\n      long value \u003d 0;\n\n      bpf_lookup_elem(fd, key, values);\n      for (i \u003d 0; i \u003c nr_cpus; i++)\n        value +\u003d values[i];\n\n    The user space must provide round_up(value_size, 8) * nr_cpus\n    array to get/set values, since kernel will use \u0027long\u0027 copy\n    of per-cpu values to try to copy good counters atomically.\n    It\u0027s a best-effort, since bpf programs and user space are racing\n    to access the same memory.\n\n    Signed-off-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit c795e49cf732a91cc4413809ade72b79be2bb632\nAuthor: Alexei Starovoitov \u003cast@fb.com\u003e\nDate:   Mon Feb 1 22:39:54 2016 -0800\n\n    bpf: introduce BPF_MAP_TYPE_PERCPU_ARRAY map\n\n    Primary use case is a histogram array of latency\n    where bpf program computes the latency of block requests or other\n    events and stores histogram of latency into array of 64 elements.\n    All cpus are constantly running, so normal increment is not accurate,\n    bpf_xadd causes cache ping-pong and this per-cpu approach allows\n    fastest collision-free counters.\n\n    Signed-off-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 04052cc31a398c54f25b6cf08eb0734a52942b9f\nAuthor: Alexei Starovoitov \u003cast@fb.com\u003e\nDate:   Mon Feb 1 22:39:53 2016 -0800\n\n    bpf: introduce BPF_MAP_TYPE_PERCPU_HASH map\n\n    Introduce BPF_MAP_TYPE_PERCPU_HASH map type which is used to do\n    accurate counters without need to use BPF_XADD instruction which turned\n    out to be too costly for high-performance network monitoring.\n    In the typical use case the \u0027key\u0027 is the flow tuple or other long\n    living object that sees a lot of events per second.\n\n    bpf_map_lookup_elem() returns per-cpu area.\n    Example:\n    struct {\n      u32 packets;\n      u32 bytes;\n    } * ptr \u003d bpf_map_lookup_elem(\u0026map, \u0026key);\n    /* ptr points to this_cpu area of the value, so the following\n     * increments will not collide with other cpus\n     */\n    ptr-\u003epackets ++;\n    ptr-\u003ebytes +\u003d skb-\u003elen;\n\n    bpf_update_elem() atomically creates a new element where all per-cpu\n    values are zero initialized and this_cpu value is populated with\n    given \u0027value\u0027.\n    Note that non-per-cpu hash map always allocates new element\n    and then deletes old after rcu grace period to maintain atomicity\n    of update. Per-cpu hash map updates element values in-place.\n\n    Signed-off-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 7b4e308a152a3754f5eb52eec2996b44801dde1c\nAuthor: Alexei Starovoitov \u003calexei.starovoitov@gmail.com\u003e\nDate:   Mon Jan 25 20:59:49 2016 -0800\n\n    perf/bpf: Convert perf_event_array to use struct file\n\n    Robustify refcounting.\n\n    Signed-off-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Signed-off-by: Peter Zijlstra (Intel) \u003cpeterz@infradead.org\u003e\n    Cc: Alexander Shishkin \u003calexander.shishkin@linux.intel.com\u003e\n    Cc: Arnaldo Carvalho de Melo \u003cacme@infradead.org\u003e\n    Cc: Arnaldo Carvalho de Melo \u003cacme@redhat.com\u003e\n    Cc: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Cc: David Ahern \u003cdsahern@gmail.com\u003e\n    Cc: Jiri Olsa \u003cjolsa@kernel.org\u003e\n    Cc: Jiri Olsa \u003cjolsa@redhat.com\u003e\n    Cc: Linus Torvalds \u003ctorvalds@linux-foundation.org\u003e\n    Cc: Namhyung Kim \u003cnamhyung@kernel.org\u003e\n    Cc: Peter Zijlstra \u003cpeterz@infradead.org\u003e\n    Cc: Stephane Eranian \u003ceranian@google.com\u003e\n    Cc: Thomas Gleixner \u003ctglx@linutronix.de\u003e\n    Cc: Vince Weaver \u003cvincent.weaver@maine.edu\u003e\n    Cc: Wang Nan \u003cwangnan0@huawei.com\u003e\n    Cc: vince@deater.net\n    Link: http://lkml.kernel.org/r/20160126045947.GA40151@ast-mbp.thefacebook.com\n    Signed-off-by: Ingo Molnar \u003cmingo@kernel.org\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 44d7aa703ab5024b0ddb9493f1a0aecbd2046213\nAuthor: Rabin Vincent \u003crabin@rab.in\u003e\nDate:   Tue Jan 12 20:17:08 2016 +0100\n\n    net: bpf: reject invalid shifts\n\n    On ARM64, a BUG() is triggered in the eBPF JIT if a filter with a\n    constant shift that can\u0027t be encoded in the immediate field of the\n    UBFM/SBFM instructions is passed to the JIT.  Since these shifts\n    amounts, which are negative or \u003e\u003d regsize, are invalid, reject them in\n    the eBPF verifier and the classic BPF filter checker, for all\n    architectures.\n\n    Signed-off-by: Rabin Vincent \u003crabin@rab.in\u003e\n    Acked-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Acked-by: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 5d82c86f0a0fd5da784ee9c1b16e404ae8323918\nAuthor: tom.leiming@gmail.com \u003ctom.leiming@gmail.com\u003e\nDate:   Tue Dec 29 22:40:27 2015 +0800\n\n    bpf: hash: use per-bucket spinlock\n\n    Both htab_map_update_elem() and htab_map_delete_elem() can be\n    called from eBPF program, and they may be in kernel hot path,\n    so it isn\u0027t efficient to use a per-hashtable lock in this two\n    helpers.\n\n    The per-hashtable spinlock is used for protecting bucket\u0027s\n    hlist, and per-bucket lock is just enough. This patch converts\n    the per-hashtable lock into per-bucket spinlock, so that\n    contention can be decreased a lot.\n\n    Signed-off-by: Ming Lei \u003ctom.leiming@gmail.com\u003e\n    Acked-by: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 65a9b4946bbf2e1e330a9aa6c5e51c7badfb36e9\nAuthor: tom.leiming@gmail.com \u003ctom.leiming@gmail.com\u003e\nDate:   Tue Dec 29 22:40:26 2015 +0800\n\n    bpf: hash: move select_bucket() out of htab\u0027s spinlock\n\n    The spinlock is just used for protecting the per-bucket\n    hlist, so it isn\u0027t needed for selecting bucket.\n\n    Acked-by: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Signed-off-by: Ming Lei \u003ctom.leiming@gmail.com\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit c8e06496fe4a361c06396c925593f02441dca246\nAuthor: tom.leiming@gmail.com \u003ctom.leiming@gmail.com\u003e\nDate:   Tue Dec 29 22:40:25 2015 +0800\n\n    bpf: hash: use atomic count\n\n    Preparing for removing global per-hashtable lock, so\n    the counter need to be defined as aotmic_t first.\n\n    Acked-by: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Signed-off-by: Ming Lei \u003ctom.leiming@gmail.com\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 5da59e4e0bd28d85f8b94cd7e511265dd749b418\nAuthor: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\nDate:   Thu Dec 17 23:51:54 2015 +0100\n\n    bpf: move clearing of A/X into classic to eBPF migration prologue\n\n    Back in the days where eBPF (or back then \"internal BPF\" ;-\u003e) was not\n    exposed to user space, and only the classic BPF programs internally\n    translated into eBPF programs, we missed the fact that for classic BPF\n    A and X needed to be cleared. It was fixed back then via 83d5b7ef99c9\n    (\"net: filter: initialize A and X registers\"), and thus classic BPF\n    specifics were added to the eBPF interpreter core to work around it.\n\n    This added some confusion for JIT developers later on that take the\n    eBPF interpreter code as an example for deriving their JIT. F.e. in\n    f75298f5c3fe (\"s390/bpf: clear correct BPF accumulator register\"), at\n    least X could leak stack memory. Furthermore, since this is only needed\n    for classic BPF translations and not for eBPF (verifier takes care\n    that read access to regs cannot be done uninitialized), more complexity\n    is added to JITs as they need to determine whether they deal with\n    migrations or native eBPF where they can just omit clearing A/X in\n    their prologue and thus reduce image size a bit, see f.e. cde66c2d88da\n    (\"s390/bpf: Only clear A and X for converted BPF programs\"). In other\n    cases (x86, arm64), A and X is being cleared in the prologue also for\n    eBPF case, which is unnecessary.\n\n    Lets move this into the BPF migration in bpf_convert_filter() where it\n    actually belongs as long as the number of eBPF JITs are still few. It\n    can thus be done generically; allowing us to remove the quirk from\n    __bpf_prog_run() and to slightly reduce JIT image size in case of eBPF,\n    while reducing code duplication on this matter in current(/future) eBPF\n    JITs.\n\n    Signed-off-by: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Acked-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Reviewed-by: Michael Holzheu \u003cholzheu@linux.vnet.ibm.com\u003e\n    Tested-by: Michael Holzheu \u003cholzheu@linux.vnet.ibm.com\u003e\n    Cc: Zi Shen Lim \u003czlim.lnx@gmail.com\u003e\n    Cc: Yang Shi \u003cyang.shi@linaro.org\u003e\n    Acked-by: Yang Shi \u003cyang.shi@linaro.org\u003e\n    Acked-by: Zi Shen Lim \u003czlim.lnx@gmail.com\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 514b5beddd33546aa9bb320c60f6ed317a78a6bc\nAuthor: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\nDate:   Thu Dec 10 22:33:49 2015 +0100\n\n    bpf, inode: allow for rename and link ops\n\n    Add support for renaming and hard links to the fs. Most of this can be\n    implemented by using simple library operations under the same constraints\n    that we don\u0027t use a reserved name like elsewhere. Linking can be useful\n    to share/manage things like maps across subsystem users. It works within\n    the file system boundary, but is not allowed for directories.\n\n    Symbolic links are explicitly not implemented here, as it can be better\n    done already by doing bind mounts inside bpf fs to set up shared directories\n    f.e. useful when using volumes in docker containers that map a private\n    working directory into /sys/fs/bpf/ which contains itself a bind mounted\n    path from the host\u0027s /sys/fs/bpf/ mount that is shared among multiple\n    containers. For single maps instead of whole directory, hard links can\n    be easily used to do the same.\n\n    Signed-off-by: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Acked-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 9c45cc504c3d4e42abdbc78414525f892a3b5708\nAuthor: Alexei Starovoitov \u003cast@kernel.org\u003e\nDate:   Sun Nov 29 16:59:35 2015 -0800\n\n    bpf: fix allocation warnings in bpf maps and integer overflow\n\n    For large map-\u003evalue_size the user space can trigger memory allocation warnings like:\n    WARNING: CPU: 2 PID: 11122 at mm/page_alloc.c:2989\n    __alloc_pages_nodemask+0x695/0x14e0()\n    Call Trace:\n     [\u003c     inline     \u003e] __dump_stack lib/dump_stack.c:15\n     [\u003cffffffff82743b56\u003e] dump_stack+0x68/0x92 lib/dump_stack.c:50\n     [\u003cffffffff81244ec9\u003e] warn_slowpath_common+0xd9/0x140 kernel/panic.c:460\n     [\u003cffffffff812450f9\u003e] warn_slowpath_null+0x29/0x30 kernel/panic.c:493\n     [\u003c     inline     \u003e] __alloc_pages_slowpath mm/page_alloc.c:2989\n     [\u003cffffffff81554e95\u003e] __alloc_pages_nodemask+0x695/0x14e0 mm/page_alloc.c:3235\n     [\u003cffffffff816188fe\u003e] alloc_pages_current+0xee/0x340 mm/mempolicy.c:2055\n     [\u003c     inline     \u003e] alloc_pages include/linux/gfp.h:451\n     [\u003cffffffff81550706\u003e] alloc_kmem_pages+0x16/0xf0 mm/page_alloc.c:3414\n     [\u003cffffffff815a1c89\u003e] kmalloc_order+0x19/0x60 mm/slab_common.c:1007\n     [\u003cffffffff815a1cef\u003e] kmalloc_order_trace+0x1f/0xa0 mm/slab_common.c:1018\n     [\u003c     inline     \u003e] kmalloc_large include/linux/slab.h:390\n     [\u003cffffffff81627784\u003e] __kmalloc+0x234/0x250 mm/slub.c:3525\n     [\u003c     inline     \u003e] kmalloc include/linux/slab.h:463\n     [\u003c     inline     \u003e] map_update_elem kernel/bpf/syscall.c:288\n     [\u003c     inline     \u003e] SYSC_bpf kernel/bpf/syscall.c:744\n\n    To avoid never succeeding kmalloc with order \u003e\u003d MAX_ORDER check that\n    elem-\u003evalue_size and computed elem_size are within limits for both hash and\n    array type maps.\n    Also add __GFP_NOWARN to kmalloc(value_size | elem_size) to avoid OOM warnings.\n    Note kmalloc(key_size) is highly unlikely to trigger OOM, since key_size \u003c\u003d 512,\n    so keep those kmalloc-s as-is.\n\n    Large value_size can cause integer overflows in elem_size and map.pages\n    formulas, so check for that as well.\n\n    Fixes: aaac3ba95e4c (\"bpf: charge user for creation of BPF maps and programs\")\n    Reported-by: Dmitry Vyukov \u003cdvyukov@google.com\u003e\n    Signed-off-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 93e43b4d010f09485fddc91885055f90081883eb\nAuthor: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\nDate:   Mon Nov 30 13:02:56 2015 +0100\n\n    bpf, array: fix heap out-of-bounds access when updating elements\n\n    During own review but also reported by Dmitry\u0027s syzkaller [1] it has been\n    noticed that we trigger a heap out-of-bounds access on eBPF array maps\n    when updating elements. This happens with each map whose map-\u003evalue_size\n    (specified during map creation time) is not multiple of 8 bytes.\n\n    In array_map_alloc(), elem_size is round_up(attr-\u003evalue_size, 8) and\n    used to align array map slots for faster access. However, in function\n    array_map_update_elem(), we update the element as ...\n\n    memcpy(array-\u003evalue + array-\u003eelem_size * index, value, array-\u003eelem_size);\n\n    ... where we access \u0027value\u0027 out-of-bounds, since it was allocated from\n    map_update_elem() from syscall side as kmalloc(map-\u003evalue_size, GFP_USER)\n    and later on copied through copy_from_user(value, uvalue, map-\u003evalue_size).\n    Thus, up to 7 bytes, we can access out-of-bounds.\n\n    Same could happen from within an eBPF program, where in worst case we\n    access beyond an eBPF program\u0027s designated stack.\n\n    Since 1be7f75d1668 (\"bpf: enable non-root eBPF programs\") didn\u0027t hit an\n    official release yet, it only affects priviledged users.\n\n    In case of array_map_lookup_elem(), the verifier prevents eBPF programs\n    from accessing beyond map-\u003evalue_size through check_map_access(). Also\n    from syscall side map_lookup_elem() only copies map-\u003evalue_size back to\n    user, so nothing could leak.\n\n      [1] http://github.com/google/syzkaller\n\n    Fixes: 28fbcfa08d8e (\"bpf: add array type of eBPF maps\")\n    Reported-by: Dmitry Vyukov \u003cdvyukov@google.com\u003e\n    Signed-off-by: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Acked-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 5fd3bbeeb5b76baf6fb8878377a9d25cf51b8b52\nAuthor: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\nDate:   Tue Nov 24 21:28:15 2015 +0100\n\n    bpf: fix clearing on persistent program array maps\n\n    Currently, when having map file descriptors pointing to program arrays,\n    there\u0027s still the issue that we unconditionally flush program array\n    contents via bpf_fd_array_map_clear() in bpf_map_release(). This happens\n    when such a file descriptor is released and is independent of the map\u0027s\n    refcount.\n\n    Having this flush independent of the refcount is for a reason: there\n    can be arbitrary complex dependency chains among tail calls, also circular\n    ones (direct or indirect, nesting limit determined during runtime), and\n    we need to make sure that the map drops all references to eBPF programs\n    it holds, so that the map\u0027s refcount can eventually drop to zero and\n    initiate its freeing. Btw, a walk of the whole dependency graph would\n    not be possible for various reasons, one being complexity and another\n    one inconsistency, i.e. new programs can be added to parts of the graph\n    at any time, so there\u0027s no guaranteed consistent state for the time of\n    such a walk.\n\n    Now, the program array pinning itself works, but the issue is that each\n    derived file descriptor on close would nevertheless call unconditionally\n    into bpf_fd_array_map_clear(). Instead, keep track of users and postpone\n    this flush until the last reference to a user is dropped. As this only\n    concerns a subset of references (f.e. a prog array could hold a program\n    that itself has reference on the prog array holding it, etc), we need to\n    track them separately.\n\n    Short analysis on the refcounting: on map creation time usercnt will be\n    one, so there\u0027s no change in behaviour for bpf_map_release(), if unpinned.\n    If we already fail in map_create(), we are immediately freed, and no\n    file descriptor has been made public yet. In bpf_obj_pin_user(), we need\n    to probe for a possible map in bpf_fd_probe_obj() already with a usercnt\n    reference, so before we drop the reference on the fd with fdput().\n    Therefore, if actual pinning fails, we need to drop that reference again\n    in bpf_any_put(), otherwise we keep holding it. When last reference\n    drops on the inode, the bpf_any_put() in bpf_evict_inode() will take\n    care of dropping the usercnt again. In the bpf_obj_get_user() case, the\n    bpf_any_get() will grab a reference on the usercnt, still at a time when\n    we have the reference on the path. Should we later on fail to grab a new\n    file descriptor, bpf_any_put() will drop it, otherwise we hold it until\n    bpf_map_release() time.\n\n    Joint work with Alexei.\n\n    Fixes: b2197755b263 (\"bpf: add support for persistent maps/progs\")\n    Signed-off-by: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Signed-off-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 4b3f084abf47da6140d5a96d5083ec4ef7be35d1\nAuthor: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\nDate:   Thu Nov 19 11:56:22 2015 +0100\n\n    bpf: add show_fdinfo handler for maps\n\n    Add a handler for show_fdinfo() to be used by the anon-inodes\n    backend for eBPF maps, and dump the map specification there. Not\n    only useful for admins, but also it provides a minimal way to\n    compare specs from ELF vs pinned object.\n\n    Signed-off-by: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Acked-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Acked-by: Hannes Frederic Sowa \u003channes@stressinduktion.org\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit b993170dfb6532765cb1f94774ffbd5d83af94b6\nAuthor: Anay Wadhera \u003cawadhera@berkeley.edu\u003e\nDate:   Thu May 20 22:00:28 2021 -0700\n\n    Revert \"bpf: fix clearing on persistent program array maps\"\n\n    This reverts commit c9da161c6517ba12154059d3b965c2cbaf16f90f.\n\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 6735bc67368515c04c2e5338e6d5738c7b10ef39\nAuthor: Anay Wadhera \u003cawadhera@berkeley.edu\u003e\nDate:   Thu May 20 22:00:23 2021 -0700\n\n    Revert \"bpf, array: fix heap out-of-bounds access when updating elements\"\n\n    This reverts commit fbca9d2d35c6ef1b323fae75cc9545005ba25097.\n\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 609e0835b15f33be0a50025be8acd2280a6068be\nAuthor: Anay Wadhera \u003cawadhera@berkeley.edu\u003e\nDate:   Thu May 20 22:00:16 2021 -0700\n\n    Revert \"bpf: fix allocation warnings in bpf maps and integer overflow\"\n\n    This reverts commit 01b3f52157ff5a47d6d8d796f396a4b34a53c61d.\n\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 928bd46a86c51f023a10c2f54d6cd2eaf06ca212\nAuthor: Anay Wadhera \u003cawadhera@berkeley.edu\u003e\nDate:   Thu May 20 22:00:11 2021 -0700\n\n    Revert \"net: bpf: reject invalid shifts\"\n\n    This reverts commit 35987ff2eaa05d70154c5bd28ebb2b70a7d8368b.\n\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 4cf55928c5a61a7045df3e999223874a3aff80d2\nAuthor: Anay Wadhera \u003cawadhera@berkeley.edu\u003e\nDate:   Thu May 20 22:00:06 2021 -0700\n\n    Revert \"bpf: fix branch offset adjustment on backjumps after patching ctx expansion\"\n\n    This reverts commit a34f2f9f2034f7984f9529002c6fffe9cb63189d.\n\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit b92f8a7ee0fc76df59b3af27d067cf5d35a0044e\nAuthor: Anay Wadhera \u003cawadhera@berkeley.edu\u003e\nDate:   Thu May 20 22:00:00 2021 -0700\n\n    Revert \"bpf: avoid copying junk bytes in bpf_get_current_comm()\"\n\n    This reverts commit e8e43232627082328fa4016fab1960360360f167.\n\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 05182c79d6472324ff4f9bdf076ca9d80c406f67\nAuthor: Anay Wadhera \u003cawadhera@berkeley.edu\u003e\nDate:   Thu May 20 21:59:51 2021 -0700\n\n    Revert \"bpf/verifier: reject invalid LD_ABS | BPF_DW instruction\"\n\n    This reverts commit 8427d5547d0b63beb70d3858127942f828400ad2.\n\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 88d2d86f1e5d53edeb29a3d06b3b94b6320dc700\nAuthor: Anay Wadhera \u003cawadhera@berkeley.edu\u003e\nDate:   Thu May 20 21:58:52 2021 -0700\n\n    Revert \"bpf: fix double-fdput in replace_map_fd_with_map_ptr()\"\n\n    This reverts commit 608d2c3c7a046c222cae2e857cf648a9f89e772b.\n\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 0acc470ed6c52a66f6175d73f9fe54d8648f554e\nAuthor: Anay Wadhera \u003cawadhera@berkeley.edu\u003e\nDate:   Thu May 20 21:58:41 2021 -0700\n\n    Revert \"bpf: fix refcnt overflow\"\n\n    This reverts commit 3899251bdb9c2b31fc73d4cc132f52d3710101de.\n\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 82e4ad585e90f5827a80c7e1fea7d38aee371200\nAuthor: Anay Wadhera \u003cawadhera@berkeley.edu\u003e\nDate:   Thu May 20 21:58:40 2021 -0700\n\n    Revert \"bpf: fix check_map_func_compatibility logic\"\n\n    This reverts commit bb10156f572f06f3b6cadd378e5a0ab3ed8da991.\n\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 437814d65d6be82a89c7d19c2a2d1dda7109455c\nAuthor: Anay Wadhera \u003cawadhera@berkeley.edu\u003e\nDate:   Thu May 20 21:58:12 2021 -0700\n\n    Revert \"bpf: Use mount_nodev not mount_ns to mount the bpf filesystem\"\n\n    This reverts commit 5b7ea922e1754107f77d146011612f2e42600cc1.\n\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit a2d50cfa44dab239102312a80b5f69c1206a0188\nAuthor: Anay Wadhera \u003cawadhera@berkeley.edu\u003e\nDate:   Thu May 20 21:58:07 2021 -0700\n\n    Revert \"bpf, inode: disallow userns mounts\"\n\n    This reverts commit bfe951d547bf15bf1192abd20773e6603dacadf1.\n\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 069f5b2aa8f89a5556fc531958476bdd156aec61\nAuthor: Anay Wadhera \u003cawadhera@berkeley.edu\u003e\nDate:   Thu May 20 21:58:02 2021 -0700\n\n    Revert \"bpf: prevent leaking pointer via xadd on unpriviledged\"\n\n    This reverts commit 1a4f13e0a99a85c455ff2f6dc117f6f049c039fa.\n\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 654d00c0b8287f809c30507b3a08a2e394855e00\nAuthor: Anay Wadhera \u003cawadhera@berkeley.edu\u003e\nDate:   Thu May 20 21:57:55 2021 -0700\n\n    Revert \"bpf/verifier: reject BPF_ALU64|BPF_END\"\n\n    This reverts commit 2ec54b21dd7b25df0f070f1d67db2ea18987e69e.\n\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 34467768210f021f0aec7dabff5b9e0c6b182376\nAuthor: Anay Wadhera \u003cawadhera@berkeley.edu\u003e\nDate:   Thu May 20 21:57:48 2021 -0700\n\n    Revert \"bpf: don\u0027t let ldimm64 leak map addresses on unprivileged\"\n\n    This reverts commit 49630dd2e10a3b2fee0cec19feb63f08453b876f.\n\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 37ce3f871f002c3a340e2ff6560116ea55049666\nAuthor: Anay Wadhera \u003cawadhera@berkeley.edu\u003e\nDate:   Thu May 20 21:56:46 2021 -0700\n\n    Revert \"bpf: add bpf_patch_insn_single helper\"\n\n    This reverts commit 087a92287dbae61b4ee1e76d7c20c81710109422.\n\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 46c1980e529e4870fcfc95b5406b09bb279c4bcf\nAuthor: Anay Wadhera \u003cawadhera@berkeley.edu\u003e\nDate:   Thu May 20 21:56:46 2021 -0700\n\n    Revert \"bpf: don\u0027t (ab)use instructions to store state\"\n\n    This reverts commit 0748b80e432584502d1559b1a51b7df58f5e2fce.\n\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit e4d6ebc747526631101b7e1112686b2189f4104e\nAuthor: Anay Wadhera \u003cawadhera@berkeley.edu\u003e\nDate:   Thu May 20 21:56:45 2021 -0700\n\n    Revert \"bpf: move fixup_bpf_calls() function\"\n\n    This reverts commit 14c7c55f452740549d561e583714b700cd88883e.\n\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit d7a869af8d0aa0d5436e2c8859a58ca746e6548a\nAuthor: Anay Wadhera \u003cawadhera@berkeley.edu\u003e\nDate:   Thu May 20 21:56:45 2021 -0700\n\n    Revert \"bpf: refactor fixup_bpf_calls()\"\n\n    This reverts commit 19614eee0644a59a8ea2509a6fbc0e771644a4f2.\n\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit a99ed9ef738b420625a3c9e8441043f672488bf2\nAuthor: Anay Wadhera \u003cawadhera@berkeley.edu\u003e\nDate:   Thu May 20 21:56:44 2021 -0700\n\n    Revert \"bpf: adjust insn_aux_data when patching insns\"\n\n    This reverts commit 648064515d0d91d10d255ab1e3afa3ecffc2943a.\n\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 4f5a8642bc49946f822dd2162bbaeadd64fbf5ee\nAuthor: Anay Wadhera \u003cawadhera@berkeley.edu\u003e\nDate:   Thu May 20 21:56:43 2021 -0700\n\n    Revert \"bpf: prevent out-of-bounds speculation\"\n\n    This reverts commit 9a7fad4c0e215fb1c256fee27c45f9f8bc4364c5.\n\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 8efde2b259578f829b95a1ce90b0a1261624eebc\nAuthor: Anay Wadhera \u003cawadhera@berkeley.edu\u003e\nDate:   Thu May 20 21:56:42 2021 -0700\n\n    Revert \"bpf, array: fix overflow in max_entries and undefined behavior in index_mask\"\n\n    This reverts commit 095b0ba360ff9a86c592c1293602d42a9297e047.\n\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 96569a03ca9dfbbcbdbfc93fce7b80bba423f2b9\nAuthor: Anay Wadhera \u003cawadhera@berkeley.edu\u003e\nDate:   Thu May 20 21:56:29 2021 -0700\n\n    Revert \"bpf: fix branch pruning logic\"\n\n    This reverts commit 1367d854b97493bfb1f3d24cf89ba60cb7f059ea.\n\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit da51eafe038dbda178b1435530faf663c7003db9\nAuthor: Anay Wadhera \u003cawadhera@berkeley.edu\u003e\nDate:   Thu May 20 21:56:22 2021 -0700\n\n    Revert \"bpf: fix bpf_tail_call() x64 JIT\"\n\n    This reverts commit 361fb0481247bea4da3eb122e685c8b72ef7c8a9.\n\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 857717ca02275bedfdfce1574bc2fd3c44486135\nAuthor: Anay Wadhera \u003cawadhera@berkeley.edu\u003e\nDate:   Thu May 20 21:55:49 2021 -0700\n\n    Revert \"bpf: introduce BPF_JIT_ALWAYS_ON config\"\n\n    This reverts commit 28c486744e6de4d882a1d853aa63d99fcba4b7a6.\n\n    Change-Id: Iffebc366a5c2cc47b16e7a09438b018485facb95\n\ncommit df9200d0917b01ae70a79b34ba483574718d5638\nAuthor: Anay Wadhera \u003cawadhera@berkeley.edu\u003e\nDate:   Thu May 20 21:55:29 2021 -0700\n\n    Revert \"bpf: arsh is not supported in 32 bit alu thus reject it\"\n\n    This reverts commit 7dcda40e52ff0712a2d7d5353c1722cb1f994330.\n\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 056335ef4d9ae9aa22757f2883848fe34241c59d\nAuthor: Anay Wadhera \u003cawadhera@berkeley.edu\u003e\nDate:   Thu May 20 21:55:29 2021 -0700\n\n    Revert \"bpf: avoid false sharing of map refcount with max_entries\"\n\n    This reverts commit 96d9b2338bed553c37f759127d8d18c857449ceb.\n\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 1c238b1a8e9a451be0764abc5e1b7cd0efea6afd\nAuthor: Anay Wadhera \u003cawadhera@berkeley.edu\u003e\nDate:   Thu May 20 21:55:28 2021 -0700\n\n    Revert \"bpf: fix divides by zero\"\n\n    This reverts commit b72ba2a0d82447538c7c977ccb3f2b31b19b7767.\n\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 447a58b830cce8c59d997de357f91101a60d5000\nAuthor: Anay Wadhera \u003cawadhera@berkeley.edu\u003e\nDate:   Thu May 20 21:55:27 2021 -0700\n\n    Revert \"bpf: fix 32-bit divide by zero\"\n\n    This reverts commit 02662601a231f8721930168ce71d84bcfb8d9a96.\n\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit ca6816795eebb210375fabb5566f22be9379b301\nAuthor: Anay Wadhera \u003cawadhera@berkeley.edu\u003e\nDate:   Thu May 20 21:55:26 2021 -0700\n\n    Revert \"bpf: reject stores into ctx via st and xadd\"\n\n    This reverts commit faa74a862a9442233bff39a496013a74775fb660.\n\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 0377b7697f26620e44a05d6cf49d5f9023ebd99c\nAuthor: Anay Wadhera \u003cawadhera@berkeley.edu\u003e\nDate:   Thu May 20 21:50:08 2021 -0700\n\n    Revert \"bpf: fix incorrect sign extension in check_alu_op()\"\n\n    This reverts commit a6132276ab5dcc38b3299082efeb25b948263adb.\n\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 0a4be322061ab87402dd9c59f45629f350a72be6\nAuthor: Anay Wadhera \u003cawadhera@berkeley.edu\u003e\nDate:   Thu May 20 21:50:03 2021 -0700\n\n    Revert \"bpf: skip unnecessary capability check\"\n\n    This reverts commit c9ea2f8af67399904fe9c72ab5192a0c0ae7f2bf.\n\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 5e042bd8e7f4485b185085f3a65a07db01d6d2b8\nAuthor: Anay Wadhera \u003cawadhera@berkeley.edu\u003e\nDate:   Thu May 20 21:49:58 2021 -0700\n\n    Revert \"bpf: map_get_next_key to return first key on NULL\"\n\n    This reverts commit ea7c24c78551c8b3e6a7e9824e5ad8ba6224f5fe.\n\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 007816bf4ab563a10e55bf85abd225beab0f60ab\nAuthor: Anay Wadhera \u003cawadhera@berkeley.edu\u003e\nDate:   Thu May 20 21:49:02 2021 -0700\n\n    Revert \"bpf: fix references to free_bpf_prog_info() in comments\"\n\n    This reverts commit b23dab51e987787e358397b24831505668625b8a.\n\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit fcbcf061c0db8a369c2ec1428d45a50a42e22a4c\nAuthor: Anay Wadhera \u003cawadhera@berkeley.edu\u003e\nDate:   Thu May 20 21:48:56 2021 -0700\n\n    Revert \"bpf: generally move prog destruction to RCU deferral\"\n\n    This reverts commit e25dc63aa366fd0f61d1d9ba67b66f5d75fc4372.\n\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit c4c8e5a91e53eb7b9a0b194d3722dae8473eeaec\nAuthor: Anay Wadhera \u003cawadhera@berkeley.edu\u003e\nDate:   Thu May 20 21:48:49 2021 -0700\n\n    Revert \"bpf: support 8-byte metafield access\"\n\n    This reverts commit 3c4bb079e16e222324c68d7594b1ab6f699edfca.\n\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit b3a8f286bb9f9cf251221624f1987c0014bd927e\nAuthor: Anay Wadhera \u003cawadhera@berkeley.edu\u003e\nDate:   Thu May 20 21:48:48 2021 -0700\n\n    Revert \"bpf/verifier: Add spi variable to check_stack_write()\"\n\n    This reverts commit 168cb9b7b2839e861278f9fde03820aba32c4ee0.\n\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 813c93faf78e24fd78229f9fc1bcaf3ef07886d6\nAuthor: Anay Wadhera \u003cawadhera@berkeley.edu\u003e\nDate:   Thu May 20 21:48:47 2021 -0700\n\n    Revert \"bpf/verifier: Pass instruction index to check_mem_access() and check_xadd()\"\n\n    This reverts commit 451624d47005aace4e314b488cb70ba3ee5dcce8.\n\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 0c67c1301cc50d5a7a69c6751ba0027347a55539\nAuthor: Anay Wadhera \u003cawadhera@berkeley.edu\u003e\nDate:   Thu May 20 21:48:46 2021 -0700\n\n    Revert \"bpf: Prevent memory disambiguation attack\"\n\n    This reverts commit 1c74bd22e846b162ea6401e8d43172e0e7256ccf.\n\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 3d948c96e6babcdbf1d3bdab3dbd0935c5a9f7f3\nAuthor: Anay Wadhera \u003cawadhera@berkeley.edu\u003e\nDate:   Thu May 20 21:48:24 2021 -0700\n\n    Revert \"bpf: silence warning messages in core\"\n\n    This reverts commit 7dd2dc652435c0abb9f05ff9ef0b378fcf743f10.\n\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 33a28fdec2c5763953393b0a4d5b397f4595f655\nAuthor: Georg Veichtlbauer \u003cgeorg@vware.at\u003e\nDate:   Fri Feb 11 20:43:28 2022 +0100\n\n    Revert \"cgroup: replace __DEVEL__sane_behavior with cgroup2 fs type\"\n\n    This reverts commit cd5367ae02a488450b714a5d5f47afb014ea6541.\n\n    Change-Id: I3d43cceedfde5bf08053ba3ed806992cfc2d8523\n\ncommit 679ee5a4e643ab09c9cedef80cf5e4f61990d120\nAuthor: Stephen Smalley \u003csds@tycho.nsa.gov\u003e\nDate:   Fri Apr 8 13:52:00 2016 -0400\n\n    selinux: distinguish non-init user namespace capability checks\n\n    Distinguish capability checks against a target associated\n    with the init user namespace versus capability checks against\n    a target associated with a non-init user namespace by defining\n    and using separate security classes for the latter.\n\n    This is needed to support e.g. Chrome usage of user namespaces\n    for the Chrome sandbox without needing to allow Chrome to also\n    exercise capabilities on targets in the init user namespace.\n\n    Suggested-by: Dan Walsh \u003cdwalsh@redhat.com\u003e\n    Signed-off-by: Stephen Smalley \u003csds@tycho.nsa.gov\u003e\n    Signed-off-by: Paul Moore \u003cpaul@paul-moore.com\u003e\n    Change-Id: I6b56d3262a73dd8a410785a51e5048aab2c5e254\n\nChange-Id: Iab6c63ba26730279d2699057ea1747d52e2a20fa\n","web_links":[{"name":"GitHub","tooltip":"Open in GitWeb","url":"https://github.com/LineageOS/android_kernel_oneplus_msm8998/commit/230f727125d0cd1fc52a69ab72e29cbac151e74a"}],"resolve_conflicts_web_links":[{"name":"GitHub","tooltip":"Open in GitWeb","url":"https://github.com/LineageOS/android_kernel_oneplus_msm8998/commit/230f727125d0cd1fc52a69ab72e29cbac151e74a"}]},"branch":"refs/heads/lineage-19.0"},"c794a655dc883b90f718519667f5a91cda4e9204":{"kind":"REWORK","_number":3,"created":"2022-02-12 08:23:26.000000000","uploader":{"_account_id":5911,"name":"Georg Veichtlbauer","email":"georg@vware.at","username":"veichtlbauer","avatars":[{"url":"https://www.gravatar.com/avatar/39db6f16bd92d063f8a1762ba2009d16.jpg?d\u003didenticon\u0026r\u003dpg\u0026s\u003d32","height":32},{"url":"https://www.gravatar.com/avatar/39db6f16bd92d063f8a1762ba2009d16.jpg?d\u003didenticon\u0026r\u003dpg\u0026s\u003d56","height":56},{"url":"https://www.gravatar.com/avatar/39db6f16bd92d063f8a1762ba2009d16.jpg?d\u003didenticon\u0026r\u003dpg\u0026s\u003d100","height":100},{"url":"https://www.gravatar.com/avatar/39db6f16bd92d063f8a1762ba2009d16.jpg?d\u003didenticon\u0026r\u003dpg\u0026s\u003d120","height":120}]},"ref":"refs/changes/55/323455/3","fetch":{"anonymous http":{"url":"https://github.com/LineageOS/android_kernel_oneplus_msm8998","ref":"refs/changes/55/323455/3","commands":{"Branch":"git fetch https://github.com/LineageOS/android_kernel_oneplus_msm8998 refs/changes/55/323455/3 \u0026\u0026 git checkout -b change-323455 FETCH_HEAD","Checkout":"git fetch https://github.com/LineageOS/android_kernel_oneplus_msm8998 refs/changes/55/323455/3 \u0026\u0026 git checkout FETCH_HEAD","Cherry Pick":"git fetch https://github.com/LineageOS/android_kernel_oneplus_msm8998 refs/changes/55/323455/3 \u0026\u0026 git cherry-pick FETCH_HEAD","Format Patch":"git fetch https://github.com/LineageOS/android_kernel_oneplus_msm8998 refs/changes/55/323455/3 \u0026\u0026 git format-patch -1 --stdout FETCH_HEAD","Pull":"git pull https://github.com/LineageOS/android_kernel_oneplus_msm8998 refs/changes/55/323455/3","Reset To":"git fetch https://github.com/LineageOS/android_kernel_oneplus_msm8998 refs/changes/55/323455/3 \u0026\u0026 git reset --hard FETCH_HEAD"}}},"commit":{"parents":[{"commit":"d3dee68c56d18f064801d121e6c734b4cb806b80","subject":"Merge branch \u0027google/android-4.4-p\u0027 into lineage-18.1","web_links":[{"name":"GitHub","tooltip":"Open in GitWeb","url":"https://github.com/LineageOS/android_kernel_oneplus_msm8998/commit/d3dee68c56d18f064801d121e6c734b4cb806b80"}]}],"author":{"name":"Georg Veichtlbauer","email":"georg@vware.at","date":"2022-02-12 08:22:38.000000000","tz":60},"committer":{"name":"Georg Veichtlbauer","email":"georg@vware.at","date":"2022-02-12 08:22:38.000000000","tz":60},"subject":"Backport BPF","message":"Backport BPF\n\ncommit 9eb4f77f38c33c1ad8aca6567eccc6317ae1ee88\nAuthor: Georg Veichtlbauer \u003cgeorg@vware.at\u003e\nDate:   Sat Feb 12 07:55:13 2022 +0100\n\n    oneplus5: Remove wireguard\n\n    Change-Id: Ib9d794d2fd88c9d583a1c23e3f24313546abcffc\n\ncommit 476edac7650f9a591e96aa6b985bca858665fcf4\nAuthor: Georg Veichtlbauer \u003cgeorg@vware.at\u003e\nDate:   Sat Feb 12 07:19:33 2022 +0100\n\n    oneplus5: Enable BPF\n\n    Change-Id: I3695f04c93f159ed7afce7a865aafd70a24818a3\n\ncommit 701aa5629c3d55a030d90564f465ea7bf25103f2\nAuthor: Arnd Bergmann \u003carnd@arndb.de\u003e\nDate:   Tue Sep 6 15:10:22 2016 +0200\n\n    perf, bpf: fix conditional call to bpf_overflow_handler\n\n    The newly added bpf_overflow_handler function is only built of both\n    CONFIG_EVENT_TRACING and CONFIG_BPF_SYSCALL are enabled, but the caller\n    only checks the latter:\n\n    kernel/events/core.c: In function \u0027perf_event_alloc\u0027:\n    kernel/events/core.c:9106:27: error: \u0027bpf_overflow_handler\u0027 undeclared (first use in this function)\n\n    This changes the caller so we also skip this call if CONFIG_EVENT_TRACING\n    is disabled entirely.\n\n    Signed-off-by: Arnd Bergmann \u003carnd@arndb.de\u003e\n    Fixes: aa6a5f3cb2b2 (\"perf, bpf: add perf events core support for BPF_PROG_TYPE_PERF_EVENT programs\")\n    Acked-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n\ncommit f3fa6c067b458a2304a93a7b09f88f871c45d2b7\nAuthor: Maitreya29 \u003cMaitreyapatni30@gmail.com\u003e\nDate:   Mon Aug 30 11:35:04 2021 +0530\n\n    net: adapt bpf_xdp_copy\n\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit dbc73c68055e651f04b4f9dc918000d184c4505a\nAuthor: Maitreya29 \u003cMaitreyapatni30@gmail.com\u003e\nDate:   Mon Aug 30 10:55:49 2021 +0530\n\n    {net, kernel}: Guard proc_dointvec_minmax_bpf_restricted and nuke void *priv from cpuset_fork\n\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 257943416cca4f91adb4344ae8e8d090d5358c18\nAuthor: Maitreya29 \u003cMaitreyapatni30@gmail.com\u003e\nDate:   Sun Aug 29 21:53:14 2021 +0530\n\n    Revert \"net/compat: Add missing sock updates for SCM_RIGHTS\"\n\n    This reverts commit 34c2166235171162c55ccdc2f3f77b377da76d7c.\n\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit b1861c775c31dbb8153231efe993e1995943d412\nAuthor: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\nDate:   Tue Dec 11 12:14:12 2018 +0100\n\n    bpf: fix bpf_jit_limit knob for PAGE_SIZE \u003e\u003d 64K\n\n    [ Upstream commit fdadd04931c2d7cd294dc5b2b342863f94be53a3 ]\n\n    Michael and Sandipan report:\n\n      Commit ede95a63b5 introduced a bpf_jit_limit tuneable to limit BPF\n      JIT allocations. At compile time it defaults to PAGE_SIZE * 40000,\n      and is adjusted again at init time if MODULES_VADDR is defined.\n\n      For ppc64 kernels, MODULES_VADDR isn\u0027t defined, so we\u0027re stuck with\n      the compile-time default at boot-time, which is 0x9c400000 when\n      using 64K page size. This overflows the signed 32-bit bpf_jit_limit\n      value:\n\n      root@ubuntu:/tmp# cat /proc/sys/net/core/bpf_jit_limit\n      -1673527296\n\n      and can cause various unexpected failures throughout the network\n      stack. In one case `strace dhclient eth0` reported:\n\n      setsockopt(5, SOL_SOCKET, SO_ATTACH_FILTER, {len\u003d11, filter\u003d0x105dd27f8},\n                 16) \u003d -1 ENOTSUPP (Unknown error 524)\n\n      and similar failures can be seen with tools like tcpdump. This doesn\u0027t\n      always reproduce however, and I\u0027m not sure why. The more consistent\n      failure I\u0027ve seen is an Ubuntu 18.04 KVM guest booted on a POWER9\n      host would time out on systemd/netplan configuring a virtio-net NIC\n      with no noticeable errors in the logs.\n\n    Given this and also given that in near future some architectures like\n    arm64 will have a custom area for BPF JIT image allocations we should\n    get rid of the BPF_JIT_LIMIT_DEFAULT fallback / default entirely. For\n    4.21, we have an overridable bpf_jit_alloc_exec(), bpf_jit_free_exec()\n    so therefore add another overridable bpf_jit_alloc_exec_limit() helper\n    function which returns the possible size of the memory area for deriving\n    the default heuristic in bpf_jit_charge_init().\n\n    Like bpf_jit_alloc_exec() and bpf_jit_free_exec(), the new\n    bpf_jit_alloc_exec_limit() assumes that module_alloc() is the default\n    JIT memory provider, and therefore in case archs implement their custom\n    module_alloc() we use MODULES_{END,_VADDR} for limits and otherwise for\n    vmalloc_exec() cases like on ppc64 we use VMALLOC_{END,_START}.\n\n    Additionally, for archs supporting large page sizes, we should change\n    the sysctl to be handled as long to not run into sysctl restrictions\n    in future.\n\n    Fixes: ede95a63b5e8 (\"bpf: add bpf_jit_limit knob to restrict unpriv allocations\")\n    Reported-by: Sandipan Das \u003csandipan@linux.ibm.com\u003e\n    Reported-by: Michael Roth \u003cmdroth@linux.vnet.ibm.com\u003e\n    Signed-off-by: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Tested-by: Michael Roth \u003cmdroth@linux.vnet.ibm.com\u003e\n    Signed-off-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Signed-off-by: Sasha Levin \u003csashal@kernel.org\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit c324f0800be0ded80ded114a67f5caa76ef114c2\nAuthor: Anay Wadhera \u003canay1018@gmail.com\u003e\nDate:   Sun May 23 18:55:08 2021 +0000\n\n    Revert \"cgroup: Disable IRQs while holding css_set_lock\"\n\n    This reverts commit ac7b270e91c7b0d1b1c5544532852b55177004f1.\n\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 035dbbe629eebce994bc816f8cca260c615128cf\nAuthor: Colin Cross \u003cccross@android.com\u003e\nDate:   Tue Jul 12 19:53:24 2011 -0700\n\n    cgroup: Add generic cgroup subsystem permission checks\n\n    Rather than using explicit euid \u003d\u003d 0 checks when trying to move\n    tasks into a cgroup via CFS, move permission checks into each\n    specific cgroup subsystem. If a subsystem does not specify a\n    \u0027allow_attach\u0027 handler, then we fall back to doing our checks\n    the old way.\n\n    Use the \u0027allow_attach\u0027 handler for the \u0027cpu\u0027 cgroup to allow\n    non-root processes to add arbitrary processes to a \u0027cpu\u0027 cgroup\n    if it has the CAP_SYS_NICE capability set.\n\n    This version of the patch adds a \u0027allow_attach\u0027 handler instead\n    of reusing the \u0027can_attach\u0027 handler.  If the \u0027can_attach\u0027 handler\n    is reused, a new cgroup that implements \u0027can_attach\u0027 but not\n    the permission checks could end up with no permission checks\n    at all.\n\n    Change-Id: Icfa950aa9321d1ceba362061d32dc7dfa2c64f0c\n    Original-Author: San Mehat \u003csan@google.com\u003e\n    Signed-off-by: Colin Cross \u003cccross@android.com\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 3412f28343b012eb9222a8bb12ff5c6313bf3570\nAuthor: Rom Lemarchand \u003cromlem@android.com\u003e\nDate:   Fri Nov 7 12:48:17 2014 -0800\n\n    cgroup: refactor allow_attach function into common code\n\n    move cpu_cgroup_allow_attach to a common subsys_cgroup_allow_attach.\n    This allows any process with CAP_SYS_NICE to move tasks across cgroups if\n    they use this function as their allow_attach handler.\n\n    Bug: 18260435\n    Change-Id: I6bb4933d07e889d0dc39e33b4e71320c34a2c90f\n    Signed-off-by: Rom Lemarchand \u003cromlem@android.com\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 106351ae5ca088de167086e4ad260a4479ddac8e\nAuthor: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\nDate:   Fri Jan 22 16:00:56 2021 +0100\n\n    bpf: Fix buggy rsh min/max bounds tracking\n\n    [ no upstream commit ]\n\n    Fix incorrect bounds tracking for RSH opcode. Commit f23cc643f9ba (\"bpf: fix\n    range arithmetic for bpf map access\") had a wrong assumption about min/max\n    bounds. The new dst_reg-\u003emin_value needs to be derived by right shifting the\n    max_val bounds, not min_val, and likewise new dst_reg-\u003emax_value needs to be\n    derived by right shifting the min_val bounds, not max_val. Later stable kernels\n    than 4.9 are not affected since bounds tracking was overall reworked and they\n    already track this similarly as in the fix.\n\n    Fixes: f23cc643f9ba (\"bpf: fix range arithmetic for bpf map access\")\n    Reported-by: Ryota Shiga (Flatt Security)\n    Signed-off-by: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Reviewed-by: John Fastabend \u003cjohn.fastabend@gmail.com\u003e\n    Cc: Josef Bacik \u003cjbacik@fb.com\u003e\n    Signed-off-by: Greg Kroah-Hartman \u003cgregkh@linuxfoundation.org\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 377d1c6e912a21884a005b562ff83cfc0a3e259e\nAuthor: Tejun Heo \u003ctj@kernel.org\u003e\nDate:   Wed Aug 10 11:23:44 2016 -0400\n\n    cgroup: add tracepoints for basic operations\n\n    Debugging what goes wrong with cgroup setup can get hairy.  Add\n    tracepoints for cgroup hierarchy mount, cgroup creation/destruction\n    and task migration operations for better visibility.\n\n    Signed-off-by: Tejun Heo \u003ctj@kernel.org\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 80f7323615d1038635b0f18f5cc8ba925c2dd0e5\nAuthor: Daniel Bristot de Oliveira \u003cbristot@redhat.com\u003e\nDate:   Wed Jun 22 17:28:41 2016 -0300\n\n    cgroup: Disable IRQs while holding css_set_lock\n\n    While testing the deadline scheduler + cgroup setup I hit this\n    warning.\n\n    [  132.612935] ------------[ cut here ]------------\n    [  132.612951] WARNING: CPU: 5 PID: 0 at kernel/softirq.c:150 __local_bh_enable_ip+0x6b/0x80\n    [  132.612952] Modules linked in: (a ton of modules...)\n    [  132.612981] CPU: 5 PID: 0 Comm: swapper/5 Not tainted 4.7.0-rc2 #2\n    [  132.612981] Hardware name: QEMU Standard PC (i440FX + PIIX, 1996), BIOS 1.8.2-20150714_191134- 04/01/2014\n    [  132.612982]  0000000000000086 45c8bb5effdd088b ffff88013fd43da0 ffffffff813d229e\n    [  132.612984]  0000000000000000 0000000000000000 ffff88013fd43de0 ffffffff810a652b\n    [  132.612985]  00000096811387b5 0000000000000200 ffff8800bab29d80 ffff880034c54c00\n    [  132.612986] Call Trace:\n    [  132.612987]  \u003cIRQ\u003e  [\u003cffffffff813d229e\u003e] dump_stack+0x63/0x85\n    [  132.612994]  [\u003cffffffff810a652b\u003e] __warn+0xcb/0xf0\n    [  132.612997]  [\u003cffffffff810e76a0\u003e] ? push_dl_task.part.32+0x170/0x170\n    [  132.612999]  [\u003cffffffff810a665d\u003e] warn_slowpath_null+0x1d/0x20\n    [  132.613000]  [\u003cffffffff810aba5b\u003e] __local_bh_enable_ip+0x6b/0x80\n    [  132.613008]  [\u003cffffffff817d6c8a\u003e] _raw_write_unlock_bh+0x1a/0x20\n    [  132.613010]  [\u003cffffffff817d6c9e\u003e] _raw_spin_unlock_bh+0xe/0x10\n    [  132.613015]  [\u003cffffffff811388ac\u003e] put_css_set+0x5c/0x60\n    [  132.613016]  [\u003cffffffff8113dc7f\u003e] cgroup_free+0x7f/0xa0\n    [  132.613017]  [\u003cffffffff810a3912\u003e] __put_task_struct+0x42/0x140\n    [  132.613018]  [\u003cffffffff810e776a\u003e] dl_task_timer+0xca/0x250\n    [  132.613027]  [\u003cffffffff810e76a0\u003e] ? push_dl_task.part.32+0x170/0x170\n    [  132.613030]  [\u003cffffffff8111371e\u003e] __hrtimer_run_queues+0xee/0x270\n    [  132.613031]  [\u003cffffffff81113ec8\u003e] hrtimer_interrupt+0xa8/0x190\n    [  132.613034]  [\u003cffffffff81051a58\u003e] local_apic_timer_interrupt+0x38/0x60\n    [  132.613035]  [\u003cffffffff817d9b0d\u003e] smp_apic_timer_interrupt+0x3d/0x50\n    [  132.613037]  [\u003cffffffff817d7c5c\u003e] apic_timer_interrupt+0x8c/0xa0\n    [  132.613038]  \u003cEOI\u003e  [\u003cffffffff81063466\u003e] ? native_safe_halt+0x6/0x10\n    [  132.613043]  [\u003cffffffff81037a4e\u003e] default_idle+0x1e/0xd0\n    [  132.613044]  [\u003cffffffff810381cf\u003e] arch_cpu_idle+0xf/0x20\n    [  132.613046]  [\u003cffffffff810e8fda\u003e] default_idle_call+0x2a/0x40\n    [  132.613047]  [\u003cffffffff810e92d7\u003e] cpu_startup_entry+0x2e7/0x340\n    [  132.613048]  [\u003cffffffff81050235\u003e] start_secondary+0x155/0x190\n    [  132.613049] ---[ end trace f91934d162ce9977 ]---\n\n    The warn is the spin_(lock|unlock)_bh(\u0026css_set_lock) in the interrupt\n    context. Converting the spin_lock_bh to spin_lock_irq(save) to avoid\n    this problem - and other problems of sharing a spinlock with an\n    interrupt.\n\n    Cc: Tejun Heo \u003ctj@kernel.org\u003e\n    Cc: Li Zefan \u003clizefan@huawei.com\u003e\n    Cc: Johannes Weiner \u003channes@cmpxchg.org\u003e\n    Cc: Juri Lelli \u003cjuri.lelli@arm.com\u003e\n    Cc: Steven Rostedt \u003crostedt@goodmis.org\u003e\n    Cc: cgroups@vger.kernel.org\n    Cc: stable@vger.kernel.org # 4.5+\n    Cc: linux-kernel@vger.kernel.org\n    Reviewed-by: Rik van Riel \u003criel@redhat.com\u003e\n    Reviewed-by: \"Luis Claudio R. Goncalves\" \u003clgoncalv@redhat.com\u003e\n    Signed-off-by: Daniel Bristot de Oliveira \u003cbristot@redhat.com\u003e\n    Acked-by: Zefan Li \u003clizefan@huawei.com\u003e\n    Signed-off-by: Tejun Heo \u003ctj@kernel.org\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit e7266339d2a7c2bf8311258567e2c94fa8c67d45\nAuthor: Johannes Weiner \u003channes@cmpxchg.org\u003e\nDate:   Thu Dec 6 09:06:47 2018 -0500\n\n    FROMLIST: kernel: cgroup: add poll file operation\n\n    Cgroup has a standardized poll/notification mechanism for waking all\n    pollers on all fds when a filesystem node changes.  To allow polling for\n    custom events, add a .poll callback that can override the default.\n\n    This is in preparation for pollable cgroup pressure files which have\n    per-fd trigger configurations.\n\n    Link: http://lkml.kernel.org/r/20190124211518.244221-3-surenb@google.com\n    Signed-off-by: Johannes Weiner \u003channes@cmpxchg.org\u003e\n    Signed-off-by: Suren Baghdasaryan \u003csurenb@google.com\u003e\n    Cc: Dennis Zhou \u003cdennis@kernel.org\u003e\n    Cc: Ingo Molnar \u003cmingo@redhat.com\u003e\n    Cc: Jens Axboe \u003caxboe@kernel.dk\u003e\n    Cc: Li Zefan \u003clizefan@huawei.com\u003e\n    Cc: Peter Zijlstra \u003cpeterz@infradead.org\u003e\n    Cc: Tejun Heo \u003ctj@kernel.org\u003e\n    Signed-off-by: Andrew Morton \u003cakpm@linux-foundation.org\u003e\n    Signed-off-by: Stephen Rothwell \u003csfr@canb.auug.org.au\u003e\n\n    (in linux-next: https://git.kernel.org/pub/scm/linux/kernel/git/next/linux-next.git/commit/?id\u003dc88177361203be291a49956b6c9d5ec164ea24b2)\n\n    Conflicts:\n            include/linux/cgroup-defs.h\n            kernel/cgroup.c\n\n    1. made changes in kernel/cgroup.c instead of kernel/cgroup/cgroup.c\n    2. replaced __poll_t with unsigned int\n\n    Bug: 111308141\n    Test: modified lmkd to use PSI and tested using lmkd_unit_test\n\n    Change-Id: Ie3d914197d1f150e1d83c6206865566a7cbff1b4\n    Signed-off-by: Suren Baghdasaryan \u003csurenb@google.com\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 8bd69873bfc478a7bad03090372614e210c19058\nAuthor: Tejun Heo \u003ctj@kernel.org\u003e\nDate:   Tue Dec 27 14:49:03 2016 -0500\n\n    UPSTREAM: cgroup add cftype-\u003eopen/release() callbacks\n\n    Pipe the newly added kernfs-\u003eopen/release() callbacks through cftype.\n    While at it, as cleanup operations now can be performed from\n    -\u003erelease() instead of -\u003eseq_stop(), make the latter optional.\n\n    Signed-off-by: Tejun Heo \u003ctj@kernel.org\u003e\n    Acked-by: Acked-by: Zefan Li \u003clizefan@huawei.com\u003e\n\n    (cherry picked from commit e90cbebc3fa5caea4c8bfeb0d0157a0cee53efc7)\n\n    Bug: 111308141\n    Test: modified lmkd to use PSI and tested using lmkd_unit_test\n\n    Change-Id: Iff9794cbbc2c7067c24cb2f767bbdeffa26b5180\n    Signed-off-by: Suren Baghdasaryan \u003csurenb@google.com\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 36c0757ab8938b79be3a22de057c651235b75c0e\nAuthor: Zefan Li \u003clizefan@huawei.com\u003e\nDate:   Sat May 9 11:32:10 2020 +0800\n\n    netprio_cgroup: Fix unlimited memory leak of v2 cgroups\n\n    [ Upstream commit 090e28b229af92dc5b40786ca673999d59e73056 ]\n\n    If systemd is configured to use hybrid mode which enables the use of\n    both cgroup v1 and v2, systemd will create new cgroup on both the default\n    root (v2) and netprio_cgroup hierarchy (v1) for a new session and attach\n    task to the two cgroups. If the task does some network thing then the v2\n    cgroup can never be freed after the session exited.\n\n    One of our machines ran into OOM due to this memory leak.\n\n    In the scenario described above when sk_alloc() is called\n    cgroup_sk_alloc() thought it\u0027s in v2 mode, so it stores\n    the cgroup pointer in sk-\u003esk_cgrp_data and increments\n    the cgroup refcnt, but then sock_update_netprioidx()\n    thought it\u0027s in v1 mode, so it stores netprioidx value\n    in sk-\u003esk_cgrp_data, so the cgroup refcnt will never be freed.\n\n    Currently we do the mode switch when someone writes to the ifpriomap\n    cgroup control file. The easiest fix is to also do the switch when\n    a task is attached to a new cgroup.\n\n    Fixes: bd1060a1d671 (\"sock, cgroup: add sock-\u003esk_cgroup\")\n    Reported-by: Yang Yingliang \u003cyangyingliang@huawei.com\u003e\n    Tested-by: Yang Yingliang \u003cyangyingliang@huawei.com\u003e\n    Signed-off-by: Zefan Li \u003clizefan@huawei.com\u003e\n    Acked-by: Tejun Heo \u003ctj@kernel.org\u003e\n    Signed-off-by: Jakub Kicinski \u003ckuba@kernel.org\u003e\n    Signed-off-by: Greg Kroah-Hartman \u003cgregkh@linuxfoundation.org\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 0c43bcb93968ee118bec16dc20f318508842ae60\nAuthor: Shakeel Butt \u003cshakeelb@google.com\u003e\nDate:   Mon Mar 9 22:16:05 2020 -0700\n\n    cgroup: memcg: net: do not associate sock with unrelated cgroup\n\n    [ Upstream commit e876ecc67db80dfdb8e237f71e5b43bb88ae549c ]\n\n    We are testing network memory accounting in our setup and noticed\n    inconsistent network memory usage and often unrelated cgroups network\n    usage correlates with testing workload. On further inspection, it\n    seems like mem_cgroup_sk_alloc() and cgroup_sk_alloc() are broken in\n    irq context specially for cgroup v1.\n\n    mem_cgroup_sk_alloc() and cgroup_sk_alloc() can be called in irq context\n    and kind of assumes that this can only happen from sk_clone_lock()\n    and the source sock object has already associated cgroup. However in\n    cgroup v1, where network memory accounting is opt-in, the source sock\n    can be unassociated with any cgroup and the new cloned sock can get\n    associated with unrelated interrupted cgroup.\n\n    Cgroup v2 can also suffer if the source sock object was created by\n    process in the root cgroup or if sk_alloc() is called in irq context.\n    The fix is to just do nothing in interrupt.\n\n    WARNING: Please note that about half of the TCP sockets are allocated\n    from the IRQ context, so, memory used by such sockets will not be\n    accouted by the memcg.\n\n    The stack trace of mem_cgroup_sk_alloc() from IRQ-context:\n\n    CPU: 70 PID: 12720 Comm: ssh Tainted:  5.6.0-smp-DEV #1\n    Hardware name: ...\n    Call Trace:\n     \u003cIRQ\u003e\n     dump_stack+0x57/0x75\n     mem_cgroup_sk_alloc+0xe9/0xf0\n     sk_clone_lock+0x2a7/0x420\n     inet_csk_clone_lock+0x1b/0x110\n     tcp_create_openreq_child+0x23/0x3b0\n     tcp_v6_syn_recv_sock+0x88/0x730\n     tcp_check_req+0x429/0x560\n     tcp_v6_rcv+0x72d/0xa40\n     ip6_protocol_deliver_rcu+0xc9/0x400\n     ip6_input+0x44/0xd0\n     ? ip6_protocol_deliver_rcu+0x400/0x400\n     ip6_rcv_finish+0x71/0x80\n     ipv6_rcv+0x5b/0xe0\n     ? ip6_sublist_rcv+0x2e0/0x2e0\n     process_backlog+0x108/0x1e0\n     net_rx_action+0x26b/0x460\n     __do_softirq+0x104/0x2a6\n     do_softirq_own_stack+0x2a/0x40\n     \u003c/IRQ\u003e\n     do_softirq.part.19+0x40/0x50\n     __local_bh_enable_ip+0x51/0x60\n     ip6_finish_output2+0x23d/0x520\n     ? ip6table_mangle_hook+0x55/0x160\n     __ip6_finish_output+0xa1/0x100\n     ip6_finish_output+0x30/0xd0\n     ip6_output+0x73/0x120\n     ? __ip6_finish_output+0x100/0x100\n     ip6_xmit+0x2e3/0x600\n     ? ipv6_anycast_cleanup+0x50/0x50\n     ? inet6_csk_route_socket+0x136/0x1e0\n     ? skb_free_head+0x1e/0x30\n     inet6_csk_xmit+0x95/0xf0\n     __tcp_transmit_skb+0x5b4/0xb20\n     __tcp_send_ack.part.60+0xa3/0x110\n     tcp_send_ack+0x1d/0x20\n     tcp_rcv_state_process+0xe64/0xe80\n     ? tcp_v6_connect+0x5d1/0x5f0\n     tcp_v6_do_rcv+0x1b1/0x3f0\n     ? tcp_v6_do_rcv+0x1b1/0x3f0\n     __release_sock+0x7f/0xd0\n     release_sock+0x30/0xa0\n     __inet_stream_connect+0x1c3/0x3b0\n     ? prepare_to_wait+0xb0/0xb0\n     inet_stream_connect+0x3b/0x60\n     __sys_connect+0x101/0x120\n     ? __sys_getsockopt+0x11b/0x140\n     __x64_sys_connect+0x1a/0x20\n     do_syscall_64+0x51/0x200\n     entry_SYSCALL_64_after_hwframe+0x44/0xa9\n\n    The stack trace of mem_cgroup_sk_alloc() from IRQ-context:\n    Fixes: 2d7580738345 (\"mm: memcontrol: consolidate cgroup socket tracking\")\n    Fixes: d979a39d7242 (\"cgroup: duplicate cgroup reference when cloning sockets\")\n    Signed-off-by: Shakeel Butt \u003cshakeelb@google.com\u003e\n    Reviewed-by: Roman Gushchin \u003cguro@fb.com\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Greg Kroah-Hartman \u003cgregkh@linuxfoundation.org\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit dcf92fa846a8c669298fd6d03e3384e28ad09dc1\nAuthor: Yang Yingliang \u003cyangyingliang@huawei.com\u003e\nDate:   Thu Aug 13 20:27:57 2020 +0000\n\n    cgroup: add missing skcd-\u003eno_refcnt check in cgroup_sk_clone()\n\n    Add skcd-\u003eno_refcnt check which is missed when backporting\n    ad0f75e5f57c (\"cgroup: fix cgroup_sk_alloc() for sk_clone_lock()\").\n\n    This patch is needed in stable-4.9, stable-4.14 and stable-4.19.\n\n    Signed-off-by: Yang Yingliang \u003cyangyingliang@huawei.com\u003e\n    Signed-off-by: Sasha Levin \u003csashal@kernel.org\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 679d6ccfac567e2cf4861b02f79f444014010ab3\nAuthor: Cong Wang \u003cxiyou.wangcong@gmail.com\u003e\nDate:   Thu Jul 2 11:52:56 2020 -0700\n\n    cgroup: fix cgroup_sk_alloc() for sk_clone_lock()\n\n    [ Upstream commit ad0f75e5f57ccbceec13274e1e242f2b5a6397ed ]\n\n    When we clone a socket in sk_clone_lock(), its sk_cgrp_data is\n    copied, so the cgroup refcnt must be taken too. And, unlike the\n    sk_alloc() path, sock_update_netprioidx() is not called here.\n    Therefore, it is safe and necessary to grab the cgroup refcnt\n    even when cgroup_sk_alloc is disabled.\n\n    sk_clone_lock() is in BH context anyway, the in_interrupt()\n    would terminate this function if called there. And for sk_alloc()\n    skcd-\u003eval is always zero. So it\u0027s safe to factor out the code\n    to make it more readable.\n\n    The global variable \u0027cgroup_sk_alloc_disabled\u0027 is used to determine\n    whether to take these reference counts. It is impossible to make\n    the reference counting correct unless we save this bit of information\n    in skcd-\u003eval. So, add a new bit there to record whether the socket\n    has already taken the reference counts. This obviously relies on\n    kmalloc() to align cgroup pointers to at least 4 bytes,\n    ARCH_KMALLOC_MINALIGN is certainly larger than that.\n\n    This bug seems to be introduced since the beginning, commit\n    d979a39d7242 (\"cgroup: duplicate cgroup reference when cloning sockets\")\n    tried to fix it but not compeletely. It seems not easy to trigger until\n    the recent commit 090e28b229af\n    (\"netprio_cgroup: Fix unlimited memory leak of v2 cgroups\") was merged.\n\n    Fixes: bd1060a1d671 (\"sock, cgroup: add sock-\u003esk_cgroup\")\n    Reported-by: Cameron Berkenpas \u003ccam@neo-zeon.de\u003e\n    Reported-by: Peter Geis \u003cpgwipeout@gmail.com\u003e\n    Reported-by: Lu Fengqi \u003clufq.fnst@cn.fujitsu.com\u003e\n    Reported-by: Daniël Sonck \u003cdsonck92@gmail.com\u003e\n    Reported-by: Zhang Qiang \u003cqiang.zhang@windriver.com\u003e\n    Tested-by: Cameron Berkenpas \u003ccam@neo-zeon.de\u003e\n    Tested-by: Peter Geis \u003cpgwipeout@gmail.com\u003e\n    Tested-by: Thomas Lamprecht \u003ct.lamprecht@proxmox.com\u003e\n    Cc: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Cc: Zefan Li \u003clizefan@huawei.com\u003e\n    Cc: Tejun Heo \u003ctj@kernel.org\u003e\n    Cc: Roman Gushchin \u003cguro@fb.com\u003e\n    Signed-off-by: Cong Wang \u003cxiyou.wangcong@gmail.com\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Greg Kroah-Hartman \u003cgregkh@linuxfoundation.org\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 00fbaff8c9243f59fa9b8133af084eada44f859e\nAuthor: Chenbo Feng \u003cfengc@google.com\u003e\nDate:   Wed Mar 22 17:27:35 2017 -0700\n\n    BACKPORT: UPSTREAM: Add a eBPF helper function to retrieve socket uid\n\n    Cherry-pick from commit 6acc5c2910689fc6ee181bf63085c5efff6a42bd\n\n    Returns the owner uid of the socket inside a sk_buff. This is useful to\n    perform per-UID accounting of network traffic or per-UID packet\n    filtering. The socket need to be a fullsock otherwise overflowuid is\n    returned.\n\n    Signed-off-by: Chenbo Feng \u003cfengc@google.com\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Bug: 30950746\n    Change-Id: Idc00947ccfdd4e9f2214ffc4178d701cd9ead0ac\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit dca320d354abab90b1f168f7eae2d7095f8922e8\nAuthor: Chenbo Feng \u003cfengc@google.com\u003e\nDate:   Wed Mar 22 17:27:34 2017 -0700\n\n    BACKPORT: UPSTREAM: Add a helper function to get socket cookie in eBPF\n\n    Cherrypick from commit: 91b8270f2a4d1d9b268de90451cdca63a70052d6\n\n    Retrieve the socket cookie generated by sock_gen_cookie() from a sk_buff\n    with a known socket. Generates a new cookie if one was not yet set.If\n    the socket pointer inside sk_buff is NULL, 0 is returned. The helper\n    function coud be useful in monitoring per socket networking traffic\n    statistics and provide a unique socket identifier per namespace.\n\n    Acked-by: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Acked-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Acked-by: Willem de Bruijn \u003cwillemb@google.com\u003e\n    Signed-off-by: Chenbo Feng \u003cfengc@google.com\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Bug: 30950746\n    Change-Id: I95918dcc3ceffb3061495a859d28aee88e3cde3c\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 3b1540862e0e1d9b6a7414ad929af2bf318db82b\nAuthor: Chenbo Feng \u003cfengc@google.com\u003e\nDate:   Wed May 3 15:22:42 2017 -0700\n\n    ANDROID: Fix missing uapi headers\n\n    Update the missing bpf helper function name in bpf_func_id to keep the\n    uapi header consistent with upstream uapi header because we need the\n    new added bpf helper function bpf get_socket_cookie and get_socket_uid.\n    The patch related to those headers are not backetported since they are\n    not related and backport them will bring in extra confilict.\n\n    Signed-off-by: Chenbo Feng \u003cfengc@google.com\u003e\n    Bug: 30950746\n    Change-Id: I2b5fd03799ac5f2e3243ab11a1bccb932f06c312\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 501b46fc55f7c7e775bb0781840b97ac941abc0f\nAuthor: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\nDate:   Fri Sep 23 01:28:37 2016 +0200\n\n    bpf: add helper to invalidate hash\n\n    Add a small helper that complements 36bbef52c7eb (\"bpf: direct packet\n    write and access for helpers for clsact progs\") for invalidating the\n    current skb-\u003ehash after mangling on headers via direct packet write.\n\n    Signed-off-by: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Acked-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit fa628ee0d7d992f894858f6caa52aed705859859\nAuthor: Anay Wadhera \u003cawadhera@berkeley.edu\u003e\nDate:   Fri May 21 15:18:35 2021 -0700\n\n    net: take compile fix from 4.9\n\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 85f8a970b122b0d14f5b9cbda702e53cb234830c\nAuthor: Anay Wadhera \u003cawadhera@berkeley.edu\u003e\nDate:   Fri May 21 15:06:45 2021 -0700\n\n    cgroup: replace out_idr_free with actual code\n\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit ea97dd71f8b3f09427f58b6a5844cb1ad622e912\nAuthor: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\nDate:   Wed Mar 9 03:00:02 2016 +0100\n\n    ip_tunnel: add support for setting flow label via collect metadata\n\n    This patch extends udp_tunnel6_xmit_skb() to pass in the IPv6 flow label\n    from call sites. Currently, there\u0027s no such option and it\u0027s always set to\n    zero when writing ip6_flow_hdr(). Add a label member to ip_tunnel_key, so\n    that flow-based tunnels via collect metadata frontends can make use of it.\n    vxlan and geneve will be converted to add flow label support separately.\n\n    Signed-off-by: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 1698a59703752faac09a9d2eb356a678c7ceb153\nAuthor: Jamal Hadi Salim \u003cjhs@mojatatu.com\u003e\nDate:   Sat Jul 2 06:43:14 2016 -0400\n\n    net: simplify and make pkt_type_ok() available for other users\n\n    Suggested-by: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Signed-off-by: Jamal Hadi Salim \u003cjhs@mojatatu.com\u003e\n    Acked-by: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit c7ac201520d2e7a760e1d4c7d11a4335ddf50178\nAuthor: Aditya Kali \u003cadityakali@google.com\u003e\nDate:   Fri Jan 29 02:54:08 2016 -0600\n\n    kernfs: define kernfs_node_dentry\n\n    Add a new kernfs api is added to lookup the dentry for a particular\n    kernfs path.\n\n    Signed-off-by: Aditya Kali \u003cadityakali@google.com\u003e\n    Signed-off-by: Serge E. Hallyn \u003cserge.hallyn@canonical.com\u003e\n    Acked-by: Greg Kroah-Hartman \u003cgregkh@linuxfoundation.org\u003e\n    Signed-off-by: Tejun Heo \u003ctj@kernel.org\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 46eb6ea46006a1f7c8b6c17ef2fdeff5f06f1232\nAuthor: Tejun Heo \u003ctj@kernel.org\u003e\nDate:   Fri Aug 11 05:49:01 2017 -0700\n\n    BACKPORT: cgroup: misc changes\n\n    Misc trivial changes to prepare for future changes.  No functional\n    difference.\n\n    * Expose cgroup_get(), cgroup_tryget() and cgroup_parent().\n\n    * Implement task_dfl_cgroup() which dereferences css_set-\u003edfl_cgrp.\n\n    * Rename cgroup_stats_show() to cgroup_stat_show() for consistency\n      with the file name.\n\n    Signed-off-by: Tejun Heo \u003ctj@kernel.org\u003e\n\n    (cherry picked from commit 3e48930cc74f0c212ee1838f89ad0ca7fcf2fea1)\n\n    Conflicts:\n            kernel/cgroup/cgroup.c\n\n    (1. manual merge because kernel/cgroup/cgroup.c is under kernel/cgroup.c\n    2. cgroup_stats_show change is skipped because the function dos not exist)\n\n    Bug: 111308141\n    Test: modified lmkd to use PSI and tested using lmkd_unit_test\n\n    Signed-off-by: Suren Baghdasaryan \u003csurenb@google.com\u003e\n    Change-Id: I756ee3dcf0d0f3da69cd1b58e644271625053538\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit b2920afd69e20d7f6fa98af3552a6d3ada8ddd62\nAuthor: Josh Poimboeuf \u003cjpoimboe@redhat.com\u003e\nDate:   Wed Mar 1 12:04:44 2017 -0600\n\n    objtool, modules: Discard objtool annotation sections for modules\n\n    commit e390f9a9689a42f477a6073e2e7df530a4c1b740 upstream.\n\n    The \u0027__unreachable\u0027 and \u0027__func_stack_frame_non_standard\u0027 sections are\n    only used at compile time.  They\u0027re discarded for vmlinux but they\n    should also be discarded for modules.\n\n    Since this is a recurring pattern, prefix the section names with\n    \".discard.\".  It\u0027s a nice convention and vmlinux.lds.h already discards\n    such sections.\n\n    Also remove the \u0027a\u0027 (allocatable) flag from the __unreachable section\n    since it doesn\u0027t make sense for a discarded section.\n\n    Suggested-by: Linus Torvalds \u003ctorvalds@linux-foundation.org\u003e\n    Signed-off-by: Josh Poimboeuf \u003cjpoimboe@redhat.com\u003e\n    Cc: Jessica Yu \u003cjeyu@redhat.com\u003e\n    Cc: Peter Zijlstra \u003cpeterz@infradead.org\u003e\n    Cc: Thomas Gleixner \u003ctglx@linutronix.de\u003e\n    Fixes: d1091c7fa3d5 (\"objtool: Improve detection of BUG() and other dead ends\")\n    Link: http://lkml.kernel.org/r/20170301180444.lhd53c5tibc4ns77@treble\n    Signed-off-by: Ingo Molnar \u003cmingo@kernel.org\u003e\n    [dwmw2: Remove the unreachable part in backporting since it\u0027s not here yet]\n    Signed-off-by: David Woodhouse \u003cdwmw@amazon.co.ku\u003e\n    Signed-off-by: Greg Kroah-Hartman \u003cgregkh@linuxfoundation.org\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit e3ac0da972457c69b5bf1c12cea73895bddf74c2\nAuthor: Josh Poimboeuf \u003cjpoimboe@redhat.com\u003e\nDate:   Sun Feb 28 22:22:35 2016 -0600\n\n    objtool: Add STACK_FRAME_NON_STANDARD() macro\n\n    Add a new macro, STACK_FRAME_NON_STANDARD(), which is used to denote a\n    function which does something unusual related to its stack frame.  Use\n    of the macro prevents objtool from emitting a false positive warning.\n\n    Signed-off-by: Josh Poimboeuf \u003cjpoimboe@redhat.com\u003e\n    Cc: Andrew Morton \u003cakpm@linux-foundation.org\u003e\n    Cc: Andy Lutomirski \u003cluto@kernel.org\u003e\n    Cc: Arnaldo Carvalho de Melo \u003cacme@kernel.org\u003e\n    Cc: Bernd Petrovitsch \u003cbernd@petrovitsch.priv.at\u003e\n    Cc: Borislav Petkov \u003cbp@alien8.de\u003e\n    Cc: Chris J Arges \u003cchris.j.arges@canonical.com\u003e\n    Cc: Jiri Slaby \u003cjslaby@suse.cz\u003e\n    Cc: Linus Torvalds \u003ctorvalds@linux-foundation.org\u003e\n    Cc: Michal Marek \u003cmmarek@suse.cz\u003e\n    Cc: Namhyung Kim \u003cnamhyung@gmail.com\u003e\n    Cc: Pedro Alves \u003cpalves@redhat.com\u003e\n    Cc: Peter Zijlstra \u003cpeterz@infradead.org\u003e\n    Cc: Thomas Gleixner \u003ctglx@linutronix.de\u003e\n    Cc: live-patching@vger.kernel.org\n    Link: http://lkml.kernel.org/r/34487a17b23dba43c50941599d47054a9584b219.1456719558.git.jpoimboe@redhat.com\n    Signed-off-by: Ingo Molnar \u003cmingo@kernel.org\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 6adc16f795a7927205b75be6599d8716873247bd\nAuthor: Anay Wadhera \u003cawadhera@berkeley.edu\u003e\nDate:   Fri May 21 14:45:03 2021 -0700\n\n    remove leftovers from 6ea07b4590d3174a53303\n\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 8afccf1a4486aeb9fa4c181d93c1ee04156a1ac2\nAuthor: Eric W. Biederman \u003cebiederm@xmission.com\u003e\nDate:   Tue May 24 09:29:01 2016 -0500\n\n    fs: Add user namespace member to struct super_block\n\n    Start marking filesystems with a user namespace owner, s_user_ns.  In\n    this change this is only used for permission checks of who may mount a\n    filesystem.  Ultimately s_user_ns will be used for translating ids and\n    checking capabilities for filesystems mounted from user namespaces.\n\n    The default policy for setting s_user_ns is implemented in sget(),\n    which arranges for s_user_ns to be set to current_user_ns() and to\n    ensure that the mounter of the filesystem has CAP_SYS_ADMIN in that\n    user_ns.\n\n    The guts of sget are split out into another function sget_userns().\n    The function sget_userns calls alloc_super with the specified user\n    namespace or it verifies the existing superblock that was found\n    has the expected user namespace, and fails with EBUSY when it is not.\n    This failing prevents users with the wrong privileges mounting a\n    filesystem.\n\n    The reason for the split of sget_userns from sget is that in some\n    cases such as mount_ns and kernfs_mount_ns a different policy for\n    permission checking of mounts and setting s_user_ns is necessary, and\n    the existence of sget_userns() allows those policies to be\n    implemented.\n\n    The helper mount_ns is expected to be used for filesystems such as\n    proc and mqueuefs which present per namespace information.  The\n    function mount_ns is modified to call sget_userns instead of sget to\n    ensure the user namespace owner of the namespace whose information is\n    presented by the filesystem is used on the superblock.\n\n    For sysfs and cgroup the appropriate permission checks are already in\n    place, and kernfs_mount_ns is modified to call sget_userns so that\n    the init_user_ns is the only user namespace used.\n\n    For the cgroup filesystem cgroup namespace mounts are bind mounts of a\n    subset of the full cgroup filesystem and as such s_user_ns must be the\n    same for all of them as there is only a single superblock.\n\n    Mounts of sysfs that vary based on the network namespace could in principle\n    change s_user_ns but it keeps the analysis and implementation of kernfs\n    simpler if that is not supported, and at present there appear to be no\n    benefits from supporting a different s_user_ns on any sysfs mount.\n\n    Getting the details of setting s_user_ns correct has been\n    a long process.  Thanks to Pavel Tikhorirorv who spotted a leak\n    in sget_userns.  Thanks to Seth Forshee who has kept the work alive.\n\n    Thanks-to: Seth Forshee \u003cseth.forshee@canonical.com\u003e\n    Thanks-to: Pavel Tikhomirov \u003cptikhomirov@virtuozzo.com\u003e\n    Acked-by: Seth Forshee \u003cseth.forshee@canonical.com\u003e\n    Signed-off-by: Eric W. Biederman \u003cebiederm@xmission.com\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 86a67faa2faee2794df97f6de68356019e5e90d1\nAuthor: Eric W. Biederman \u003cebiederm@xmission.com\u003e\nDate:   Mon May 23 14:51:59 2016 -0500\n\n    vfs: Pass data, ns, and ns-\u003euserns to mount_ns\n\n    Today what is normally called data (the mount options) is not passed\n    to fill_super through mount_ns.\n\n    Pass the mount options and the namespace separately to mount_ns so\n    that filesystems such as proc that have mount options, can use\n    mount_ns.\n\n    Pass the user namespace to mount_ns so that the standard permission\n    check that verifies the mounter has permissions over the namespace can\n    be performed in mount_ns instead of in each filesystems .mount method.\n    Thus removing the duplication between mqueuefs and proc in terms of\n    permission checks.  The extra permission check does not currently\n    affect the rpc_pipefs filesystem and the nfsd filesystem as those\n    filesystems do not currently allow unprivileged mounts.  Without\n    unpvileged mounts it is guaranteed that the caller has already passed\n    capable(CAP_SYS_ADMIN) which guarantees extra permission check will\n    pass.\n\n    Update rpc_pipefs and the nfsd filesystem to ensure that the network\n    namespace reference is always taken in fill_super and always put in kill_sb\n    so that the logic is simpler and so that errors originating inside of\n    fill_super do not cause a network namespace leak.\n\n    Acked-by: Seth Forshee \u003cseth.forshee@canonical.com\u003e\n    Signed-off-by: \"Eric W. Biederman\" \u003cebiederm@xmission.com\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 881f8f56a960124560bf8e091706084d2c4bc05d\nAuthor: Tejun Heo \u003ctj@kernel.org\u003e\nDate:   Wed Aug 10 11:23:44 2016 -0400\n\n    kernfs: make kernfs_path*() behave in the style of strlcpy()\n\n    kernfs_path*() functions always return the length of the full path but\n    the path content is undefined if the length is larger than the\n    provided buffer.  This makes its behavior different from strlcpy() and\n    requires error handling in all its users even when they don\u0027t care\n    about truncation.  In addition, the implementation can actully be\n    simplified by making it behave properly in strlcpy() style.\n\n    * Update kernfs_path_from_node_locked() to always fill up the buffer\n      with path.  If the buffer is not large enough, the output is\n      truncated and terminated.\n\n    * kernfs_path() no longer needs error handling.  Make it a simple\n      inline wrapper around kernfs_path_from_node().\n\n    * sysfs_warn_dup()\u0027s use of kernfs_path() doesn\u0027t need error handling.\n      Updated accordingly.\n\n    * cgroup_path()\u0027s use of kernfs_path() updated to retain the old\n      behavior.\n\n    Signed-off-by: Tejun Heo \u003ctj@kernel.org\u003e\n    Acked-by: Greg Kroah-Hartman \u003cgregkh@linuxfoundation.org\u003e\n    Acked-by: Serge Hallyn \u003cserge.hallyn@ubuntu.com\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit ded8df72129ad8fb3221f801e2c28b4b6ba86380\nAuthor: Serge Hallyn \u003cserge.hallyn@ubuntu.com\u003e\nDate:   Sun Apr 17 15:04:31 2016 -0500\n\n    kernfs_path_from_node_locked: don\u0027t overwrite nlen\n\n    We\u0027ve calculated @len to be the bytes we need for \u0027/..\u0027 entries from\n    @kn_from to the common ancestor, and calculated @nlen to be the extra\n    bytes we need to get from the common ancestor to @kn_to.  We use them\n    as such at the end.  But in the loop copying the actual entries, we\n    overwrite @nlen.  Use a temporary variable for that instead.\n\n    Without this, the return length, when the buffer is large enough, is\n    wrong.  (When the buffer is NULL or too small, the returned value is\n    correct. The buffer contents are also correct.)\n\n    Interestingly, no callers of this function are affected by this as of\n    yet.  However the upcoming cgroup_show_path() will be.\n\n    Signed-off-by: Serge Hallyn \u003cserge.hallyn@ubuntu.com\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit d95d06d1f77e2ca0de3146070bf126dc41d74947\nAuthor: Anay Wadhera \u003cawadhera@berkeley.edu\u003e\nDate:   Fri May 21 14:29:50 2021 -0700\n\n    arm64: bpf_jit_comp: drop artifact\n\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit b2502c663a5c37cbbe33f1a955fc1d3aa8367a26\nAuthor: Arnd Bergmann \u003carnd@arndb.de\u003e\nDate:   Tue Jan 10 13:08:06 2017 +0100\n\n    UPSTREAM: cgroup: move CONFIG_SOCK_CGROUP_DATA to init/Kconfig\n\n    We now \u0027select SOCK_CGROUP_DATA\u0027 but Kconfig complains that this is\n    not right when CONFIG_NET is disabled and there is no socket interface:\n\n    warning: (CGROUP_BPF) selects SOCK_CGROUP_DATA which has unmet direct dependencies (NET)\n\n    I don\u0027t know what the correct solution for this is, but simply removing\n    the dependency on NET from SOCK_CGROUP_DATA by moving it out of the\n    \u0027if NET\u0027 section avoids the warning and does not produce other build\n    errors.\n\n    Fixes: 483c4933ea09 (\"cgroup: Fix CGROUP_BPF config\")\n    Signed-off-by: Arnd Bergmann \u003carnd@arndb.de\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n\n    Fixes: Change-Id: Ib41ef78fba02eb9e592558ddbf06f9ec0aa337b6\n           (\"UPSTREAM: cgroup: Fix CGROUP_BPF config\")\n    (cherry picked from commit 73b351473547e543e9c8166dd67fd99c64c15b0b)\n    Signed-off-by: Amit Pundir \u003camit.pundir@linaro.org\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 3524e92fd265f0a60653aacba5f736ef7364f750\nAuthor: Andy Lutomirski \u003cluto@kernel.org\u003e\nDate:   Fri Dec 16 08:33:45 2016 -0800\n\n    UPSTREAM: cgroup: Fix CGROUP_BPF config\n\n    Cherry-pick from commit 483c4933ea09b7aa625b9d64af286fc22ec7e419\n\n    CGROUP_BPF depended on SOCK_CGROUP_DATA which can\u0027t be manually\n    enabled, making it rather challenging to turn CGROUP_BPF on.\n\n    Signed-off-by: Andy Lutomirski \u003cluto@kernel.org\u003e\n    Acked-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Bug: 30950746\n    Change-Id: Ib41ef78fba02eb9e592558ddbf06f9ec0aa337b6\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 3d5711394a7048c2effcf38b9e333768384bcc77\nAuthor: Yonghong Song \u003cyhs@fb.com\u003e\nDate:   Mon Oct 23 23:53:08 2017 -0700\n\n    BACKPORT: bpf: permit multiple bpf attachments for a single perf event\n\n    This patch enables multiple bpf attachments for a\n    kprobe/uprobe/tracepoint single trace event.\n    Each trace_event keeps a list of attached perf events.\n    When an event happens, all attached bpf programs will\n    be executed based on the order of attachment.\n\n    A global bpf_event_mutex lock is introduced to protect\n    prog_array attaching and detaching. An alternative will\n    be introduce a mutex lock in every trace_event_call\n    structure, but it takes a lot of extra memory.\n    So a global bpf_event_mutex lock is a good compromise.\n\n    The bpf prog detachment involves allocation of memory.\n    If the allocation fails, a dummy do-nothing program\n    will replace to-be-detached program in-place.\n\n    Signed-off-by: Yonghong Song \u003cyhs@fb.com\u003e\n    Acked-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Acked-by: Martin KaFai Lau \u003ckafai@fb.com\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    (cherry picked from commit e87c6bc3852b981e71c757be20771546ce9f76f3)\n    Signed-off-by: Connor O\u0027Brien \u003cconnoro@google.com\u003e\n    Bug: 121213201\n    Bug: 138317270\n    Test: build \u0026 boot cuttlefish; attach 2 progs to 1 tracepoint\n    Change-Id: I390d8c0146888ddb1aed5a6f6e5dae7ef394ebc9\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit a763ad431db7f4d7acf21760a2b0ff568d3f2a18\nAuthor: Alexei Starovoitov \u003cast@fb.com\u003e\nDate:   Mon Apr 18 20:11:50 2016 -0700\n\n    perf, bpf: minimize the size of perf_trace_() tracepoint handler\n\n    move trace_call_bpf() into helper function to minimize the size\n    of perf_trace_*() tracepoint handlers.\n        text\t   data\t    bss\t    dec\t \t   hex\tfilename\n    10541679\t5526646\t2945024\t19013349\t1221ee5\tvmlinux_before\n    10509422\t5526646\t2945024\t18981092\t121a0e4\tvmlinux_after\n\n    It may seem that perf_fetch_caller_regs() can also be moved,\n    but that is incorrect, since ip/sp will be wrong.\n\n    bpf+tracepoint performance is not affected, since\n    perf_swevent_put_recursion_context() is now inlined.\n    export_symbol_gpl can also be dropped.\n\n    No measurable change in normal perf tracepoints.\n\n    Suggested-by: Steven Rostedt \u003crostedt@goodmis.org\u003e\n    Signed-off-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Acked-by: Peter Zijlstra (Intel) \u003cpeterz@infradead.org\u003e\n    Acked-by: Steven Rostedt \u003crostedt@goodmis.org\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit a47dd93e2e80c7a62611c261411181c49412d611\nAuthor: Yonghong Song \u003cyhs@fb.com\u003e\nDate:   Mon Oct 23 23:53:07 2017 -0700\n\n    UPSTREAM: bpf: use the same condition in perf event set/free bpf handler\n\n    This is a cleanup such that doing the same check in\n    perf_event_free_bpf_prog as we already do in\n    perf_event_set_bpf_prog step.\n\n    Signed-off-by: Yonghong Song \u003cyhs@fb.com\u003e\n    Acked-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Acked-by: Martin KaFai Lau \u003ckafai@fb.com\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    (cherry picked from commit 0b4c6841fee03e096b735074a0c4aab3a8e92986)\n    Signed-off-by: Connor O\u0027Brien \u003cconnoro@google.com\u003e\n    Bug: 121213201\n    Bug: 138317270\n    Test: build \u0026 boot cuttlefish\n    Change-Id: Ie423e73a73be29e8ef50cc22dbb03e14e241c8de\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 19e3aef3398305832491b816e2374a9734ebe5a9\nAuthor: Alexei Starovoitov \u003cast@fb.com\u003e\nDate:   Mon Oct 2 22:50:21 2017 -0700\n\n    BACKPORT: bpf: multi program support for cgroup+bpf\n\n    introduce BPF_F_ALLOW_MULTI flag that can be used to attach multiple\n    bpf programs to a cgroup.\n\n    The difference between three possible flags for BPF_PROG_ATTACH command:\n    - NONE(default): No further bpf programs allowed in the subtree.\n    - BPF_F_ALLOW_OVERRIDE: If a sub-cgroup installs some bpf program,\n      the program in this cgroup yields to sub-cgroup program.\n    - BPF_F_ALLOW_MULTI: If a sub-cgroup installs some bpf program,\n      that cgroup program gets run in addition to the program in this cgroup.\n\n    NONE and BPF_F_ALLOW_OVERRIDE existed before. This patch doesn\u0027t\n    change their behavior. It only clarifies the semantics in relation\n    to new flag.\n\n    Only one program is allowed to be attached to a cgroup with\n    NONE or BPF_F_ALLOW_OVERRIDE flag.\n    Multiple programs are allowed to be attached to a cgroup with\n    BPF_F_ALLOW_MULTI flag. They are executed in FIFO order\n    (those that were attached first, run first)\n    The programs of sub-cgroup are executed first, then programs of\n    this cgroup and then programs of parent cgroup.\n    All eligible programs are executed regardless of return code from\n    earlier programs.\n\n    To allow efficient execution of multiple programs attached to a cgroup\n    and to avoid penalizing cgroups without any programs attached\n    introduce \u0027struct bpf_prog_array\u0027 which is RCU protected array\n    of pointers to bpf programs.\n\n    Signed-off-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Acked-by: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Acked-by: Martin KaFai Lau \u003ckafai@fb.com\u003e\n    for cgroup bits\n    Acked-by: Tejun Heo \u003ctj@kernel.org\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    (cherry picked from commit 324bda9e6c5add86ba2e1066476481c48132aca0)\n    Signed-off-by: Connor O\u0027Brien \u003cconnoro@google.com\u003e\n    Bug: 121213201\n    Bug: 138317270\n    Test: build \u0026 boot cuttlefish\n    Change-Id: I06b71c850b9f3e052b106abab7a4a3add012a3f8\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 774c80d9f4d227bace4c051e5176dadf6a946055\nAuthor: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\nDate:   Sat Aug 17 00:00:08 2019 +0100\n\n    bpf: add bpf_jit_limit knob to restrict unpriv allocations\n\n    commit ede95a63b5e84ddeea6b0c473b36ab8bfd8c6ce3 upstream.\n\n    Rick reported that the BPF JIT could potentially fill the entire module\n    space with BPF programs from unprivileged users which would prevent later\n    attempts to load normal kernel modules or privileged BPF programs, for\n    example. If JIT was enabled but unsuccessful to generate the image, then\n    before commit 290af86629b2 (\"bpf: introduce BPF_JIT_ALWAYS_ON config\")\n    we would always fall back to the BPF interpreter. Nowadays in the case\n    where the CONFIG_BPF_JIT_ALWAYS_ON could be set, then the load will abort\n    with a failure since the BPF interpreter was compiled out.\n\n    Add a global limit and enforce it for unprivileged users such that in case\n    of BPF interpreter compiled out we fail once the limit has been reached\n    or we fall back to BPF interpreter earlier w/o using module mem if latter\n    was compiled in. In a next step, fair share among unprivileged users can\n    be resolved in particular for the case where we would fail hard once limit\n    is reached.\n\n    Fixes: 290af86629b2 (\"bpf: introduce BPF_JIT_ALWAYS_ON config\")\n    Fixes: 0a14842f5a3c (\"net: filter: Just In Time compiler for x86-64\")\n    Co-Developed-by: Rick Edgecombe \u003crick.p.edgecombe@intel.com\u003e\n    Signed-off-by: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Acked-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Cc: Eric Dumazet \u003ceric.dumazet@gmail.com\u003e\n    Cc: Jann Horn \u003cjannh@google.com\u003e\n    Cc: Kees Cook \u003ckeescook@chromium.org\u003e\n    Cc: LKML \u003clinux-kernel@vger.kernel.org\u003e\n    Signed-off-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    [bwh: Backported to 4.9: adjust context]\n    Signed-off-by: Ben Hutchings \u003cben.hutchings@codethink.co.uk\u003e\n    Signed-off-by: Greg Kroah-Hartman \u003cgregkh@linuxfoundation.org\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 947ea352a435b1b96598dc1e6707190c73120ae9\nAuthor: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\nDate:   Fri Aug 16 23:59:56 2019 +0100\n\n    bpf: restrict access to core bpf sysctls\n\n    commit 2e4a30983b0f9b19b59e38bbf7427d7fdd480d98 upstream.\n\n    Given BPF reaches far beyond just networking these days, it was\n    never intended to allow setting and in some cases reading those\n    knobs out of a user namespace root running without CAP_SYS_ADMIN,\n    thus tighten such access.\n\n    Also the bpf_jit_enable \u003d 2 debugging mode should only be allowed\n    if kptr_restrict is not set since it otherwise can leak addresses\n    to the kernel log. Dump a note to the kernel log that this is for\n    debugging JITs only when enabled.\n\n    Signed-off-by: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Acked-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Signed-off-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    [bwh: Backported to 4.9:\n     - We don\u0027t have bpf_dump_raw_ok(), so drop the condition based on it. This\n       condition only made it a bit harder for a privileged user to do something\n       silly.\n     - Drop change to bpf_jit_kallsyms]\n    Signed-off-by: Ben Hutchings \u003cben.hutchings@codethink.co.uk\u003e\n    Signed-off-by: Greg Kroah-Hartman \u003cgregkh@linuxfoundation.org\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 4813ef2f3ac078d40bb6f94dd27163ca1d7d4c23\nAuthor: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\nDate:   Fri Aug 16 23:59:20 2019 +0100\n\n    bpf: get rid of pure_initcall dependency to enable jits\n\n    commit fa9dd599b4dae841924b022768354cfde9affecb upstream.\n\n    Having a pure_initcall() callback just to permanently enable BPF\n    JITs under CONFIG_BPF_JIT_ALWAYS_ON is unnecessary and could leave\n    a small race window in future where JIT is still disabled on boot.\n    Since we know about the setting at compilation time anyway, just\n    initialize it properly there. Also consolidate all the individual\n    bpf_jit_enable variables into a single one and move them under one\n    location. Moreover, don\u0027t allow for setting unspecified garbage\n    values on them.\n\n    Signed-off-by: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Acked-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Signed-off-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    [bwh: Backported to 4.9 as dependency of commit 2e4a30983b0f\n     \"bpf: restrict access to core bpf sysctls\":\n     - Drop change in arch/mips/net/ebpf_jit.c\n     - Drop change to bpf_jit_kallsyms\n     - Adjust filenames, context]\n    Signed-off-by: Ben Hutchings \u003cben.hutchings@codethink.co.uk\u003e\n    Signed-off-by: Greg Kroah-Hartman \u003cgregkh@linuxfoundation.org\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 6a8585e729f2bd86914235a9445885d066fce418\nAuthor: Zi Shen Lim \u003czlim.lnx@gmail.com\u003e\nDate:   Wed Jun 8 21:18:48 2016 -0700\n\n    arm64: bpf: implement bpf_tail_call() helper\n\n    Add support for JMP_CALL_X (tail call) introduced by commit 04fd61ab36ec\n    (\"bpf: allow bpf programs to tail-call other bpf programs\").\n\n    bpf_tail_call() arguments:\n      ctx   - context pointer passed to next program\n      array - pointer to map which type is BPF_MAP_TYPE_PROG_ARRAY\n      index - index inside array that selects specific program to run\n\n    In this implementation arm64 JIT jumps into callee program after prologue,\n    so callee program reuses the same stack. For tail_call_cnt, we use the\n    callee-saved R26 (which was already saved/restored but previously unused\n    by JIT).\n\n    With this patch a tail call generates the following code on arm64:\n\n      if (index \u003e\u003d array-\u003emap.max_entries)\n          goto out;\n\n      34:   mov     x10, #0x10                      // #16\n      38:   ldr     w10, [x1,x10]\n      3c:   cmp     w2, w10\n      40:   b.ge    0x0000000000000074\n\n      if (tail_call_cnt \u003e MAX_TAIL_CALL_CNT)\n          goto out;\n      tail_call_cnt++;\n\n      44:   mov     x10, #0x20                      // #32\n      48:   cmp     x26, x10\n      4c:   b.gt    0x0000000000000074\n      50:   add     x26, x26, #0x1\n\n      prog \u003d array-\u003eptrs[index];\n      if (prog \u003d\u003d NULL)\n          goto out;\n\n      54:   mov     x10, #0x68                      // #104\n      58:   ldr     x10, [x1,x10]\n      5c:   ldr     x11, [x10,x2]\n      60:   cbz     x11, 0x0000000000000074\n\n      goto *(prog-\u003ebpf_func + prologue_size);\n\n      64:   mov     x10, #0x20                      // #32\n      68:   ldr     x10, [x11,x10]\n      6c:   add     x10, x10, #0x20\n      70:   br      x10\n      74:\n\n    Signed-off-by: Zi Shen Lim \u003czlim.lnx@gmail.com\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 3c1ec8cb9b94920225f814021f141c8890ccfdad\nAuthor: Yang Shi \u003cyang.shi@linaro.org\u003e\nDate:   Mon May 16 16:36:26 2016 -0700\n\n    bpf: arm64: remove callee-save registers use for tmp registers\n\n    In the current implementation of ARM64 eBPF JIT, R23 and R24 are used for\n    tmp registers, which are callee-saved registers. This leads to variable size\n    of JIT prologue and epilogue. The latest blinding constant change prefers to\n    constant size of prologue and epilogue. AAPCS reserves R9 ~ R15 for temp\n    registers which not need to be saved/restored during function call. So, replace\n    R23 and R24 to R10 and R11, and remove tmp_used flag to save 2 instructions for\n    some jited BPF program.\n\n    CC: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Acked-by: Zi Shen Lim \u003czlim.lnx@gmail.com\u003e\n    Signed-off-by: Yang Shi \u003cyang.shi@linaro.org\u003e\n    Acked-by: Catalin Marinas \u003ccatalin.marinas@arm.com\u003e\n    Acked-by: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 76a7fa051ea364260d405238c5de63cac2d85d7c\nAuthor: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\nDate:   Fri May 13 19:08:34 2016 +0200\n\n    bpf, arm64: add support for constant blinding\n\n    This patch adds recently added constant blinding helpers into the\n    arm64 eBPF JIT. In the bpf_int_jit_compile() path, requirements are\n    to utilize bpf_jit_blind_constants()/bpf_jit_prog_release_other()\n    pair for rewriting the program into a blinded one, and to map the\n    BPF_REG_AX register to a CPU register. The mapping is on x9.\n\n    Signed-off-by: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Acked-by: Zi Shen Lim \u003czlim.lnx@gmail.com\u003e\n    Acked-by: Yang Shi \u003cyang.shi@linaro.org\u003e\n    Tested-by: Yang Shi \u003cyang.shi@linaro.org\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit cab9fa004a0766073b110d1fe6366be36359b574\nAuthor: Zi Shen Lim \u003czlim.lnx@gmail.com\u003e\nDate:   Wed Jan 13 23:33:22 2016 -0800\n\n    arm64: bpf: add extra pass to handle faulty codegen\n\n    Code generation functions in arch/arm64/kernel/insn.c previously\n    BUG_ON invalid parameters. Following change of that behavior, now we\n    need to handle the error case where AARCH64_BREAK_FAULT is returned.\n\n    Instead of error-handling on every emit() in JIT, we add a new\n    validation pass at the end of JIT compilation. There\u0027s no point in\n    running JITed code at run-time only to trap due to AARCH64_BREAK_FAULT.\n    Instead, we drop this failed JIT compilation and allow the system to\n    gracefully fallback on the BPF interpreter.\n\n    Signed-off-by: Zi Shen Lim \u003czlim.lnx@gmail.com\u003e\n    Suggested-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Acked-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 872d04a160c464374ee876f9f20d2f505d422a07\nAuthor: Valdis Klētnieks \u003cvaldis.kletnieks@vt.edu\u003e\nDate:   Thu Jun 6 22:39:27 2019 -0400\n\n    bpf: silence warning messages in core\n\n    [ Upstream commit aee450cbe482a8c2f6fa5b05b178ef8b8ff107ca ]\n\n    Compiling kernel/bpf/core.c with W\u003d1 causes a flood of warnings:\n\n    kernel/bpf/core.c:1198:65: warning: initialized field overwritten [-Woverride-init]\n     1198 | #define BPF_INSN_3_TBL(x, y, z) [BPF_##x | BPF_##y | BPF_##z] \u003d true\n          |                                                                 ^~~~\n    kernel/bpf/core.c:1087:2: note: in expansion of macro \u0027BPF_INSN_3_TBL\u0027\n     1087 |  INSN_3(ALU, ADD,  X),   \\\n          |  ^~~~~~\n    kernel/bpf/core.c:1202:3: note: in expansion of macro \u0027BPF_INSN_MAP\u0027\n     1202 |   BPF_INSN_MAP(BPF_INSN_2_TBL, BPF_INSN_3_TBL),\n          |   ^~~~~~~~~~~~\n    kernel/bpf/core.c:1198:65: note: (near initialization for \u0027public_insntable[12]\u0027)\n     1198 | #define BPF_INSN_3_TBL(x, y, z) [BPF_##x | BPF_##y | BPF_##z] \u003d true\n          |                                                                 ^~~~\n    kernel/bpf/core.c:1087:2: note: in expansion of macro \u0027BPF_INSN_3_TBL\u0027\n     1087 |  INSN_3(ALU, ADD,  X),   \\\n          |  ^~~~~~\n    kernel/bpf/core.c:1202:3: note: in expansion of macro \u0027BPF_INSN_MAP\u0027\n     1202 |   BPF_INSN_MAP(BPF_INSN_2_TBL, BPF_INSN_3_TBL),\n          |   ^~~~~~~~~~~~\n\n    98 copies of the above.\n\n    The attached patch silences the warnings, because we *know* we\u0027re overwriting\n    the default initializer. That leaves bpf/core.c with only 6 other warnings,\n    which become more visible in comparison.\n\n    Signed-off-by: Valdis Kletnieks \u003cvaldis.kletnieks@vt.edu\u003e\n    Acked-by: Andrii Nakryiko \u003candriin@fb.com\u003e\n    Signed-off-by: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Signed-off-by: Sasha Levin \u003csashal@kernel.org\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 327426de64d46c560e3327c441f719b98eb85f61\nAuthor: Chenbo Feng \u003cfengc@google.com\u003e\nDate:   Tue May 14 19:42:57 2019 -0700\n\n    UPSTREAM: bpf: relax inode permission check for retrieving bpf program\n\n    For iptable module to load a bpf program from a pinned location, it\n    only retrieve a loaded program and cannot change the program content so\n    requiring a write permission for it might not be necessary.\n    Also when adding or removing an unrelated iptable rule, it might need to\n    flush and reload the xt_bpf related rules as well and triggers the inode\n    permission check. It might be better to remove the write premission\n    check for the inode so we won\u0027t need to grant write access to all the\n    processes that flush and restore iptables rules.\n\n    Signed-off-by: Chenbo Feng \u003cfengc@google.com\u003e\n    Signed-off-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    (cherry picked from commit e547ff3f803e779a3898f1f48447b29f43c54085)\n\n    Bug: 129650054\n    Change-Id: I71487ad6f4d22e0a8be3757d9b72d1c04c92104d\n    (cherry picked from commit 9e74c1b9e8418aa0209b15db24f0b3d4876f52aa)\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 3ff714edc40981bd4a16efce329be1eefaf948fe\nAuthor: Alexei Starovoitov \u003cast@fb.com\u003e\nDate:   Thu May 9 19:33:54 2019 -0700\n\n    bpf: convert htab map to hlist_nulls\n\n    commit 4fe8435909fddc97b81472026aa954e06dd192a5 upstream.\n\n    when all map elements are pre-allocated one cpu can delete and reuse htab_elem\n    while another cpu is still walking the hlist. In such case the lookup may\n    miss the element. Convert hlist to hlist_nulls to avoid such scenario.\n    When bucket lock is taken there is no need to take such precautions,\n    so only convert map_lookup and map_get_next to nulls.\n    The race window is extremely small and only reproducible with explicit\n    udelay() inside lookup_nulls_elem_raw()\n\n    Similar to hlist add hlist_nulls_for_each_entry_safe() and\n    hlist_nulls_entry_safe() helpers.\n\n    Fixes: 6c9059817432 (\"bpf: pre-allocate hash map elements\")\n    Reported-by: Jonathan Perry \u003cjonperry@fb.com\u003e\n    Signed-off-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Acked-by: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Chenbo Feng \u003cfengc@google.com\u003e\n    Signed-off-by: Sasha Levin \u003csashal@kernel.org\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 82b83a495b0fc28c407332c29a0dd455e55bf11a\nAuthor: Alexei Starovoitov \u003cast@fb.com\u003e\nDate:   Thu May 9 19:33:53 2019 -0700\n\n    bpf: fix struct htab_elem layout\n\n    commit 9f691549f76d488a0c74397b3e51e943865ea01f upstream.\n\n    when htab_elem is removed from the bucket list the htab_elem.hash_node.next\n    field should not be overridden too early otherwise we have a tiny race window\n    between lookup and delete.\n    The bug was discovered by manual code analysis and reproducible\n    only with explicit udelay() in lookup_elem_raw().\n\n    Fixes: 6c9059817432 (\"bpf: pre-allocate hash map elements\")\n    Reported-by: Jonathan Perry \u003cjonperry@fb.com\u003e\n    Signed-off-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Acked-by: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Chenbo Feng \u003cfengc@google.com\u003e\n    Signed-off-by: Sasha Levin \u003csashal@kernel.org\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 183c5b89665f4d494194a405e2a79ab9dd917f58\nAuthor: Alexei Starovoitov \u003cast@kernel.org\u003e\nDate:   Mon Dec 3 22:46:04 2018 -0800\n\n    bpf: check pending signals while verifying programs\n\n    [ Upstream commit c3494801cd1785e2c25f1a5735fa19ddcf9665da ]\n\n    Malicious user space may try to force the verifier to use as much cpu\n    time and memory as possible. Hence check for pending signals\n    while verifying the program.\n    Note that suspend of sys_bpf(PROG_LOAD) syscall will lead to EAGAIN,\n    since the kernel has to release the resources used for program verification.\n\n    Reported-by: Anatoly Trosinenko \u003canatoly.trosinenko@gmail.com\u003e\n    Signed-off-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Acked-by: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Acked-by: Edward Cree \u003cecree@solarflare.com\u003e\n    Signed-off-by: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Signed-off-by: Sasha Levin \u003csashal@kernel.org\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 65c030b10e0404acbaf11f6d0de1edb6a06ec709\nAuthor: Alexei Starovoitov \u003cast@kernel.org\u003e\nDate:   Tue May 15 09:27:05 2018 -0700\n\n    bpf: Prevent memory disambiguation attack\n\n    commit af86ca4e3088fe5eacf2f7e58c01fa68ca067672 upstream.\n\n    Detect code patterns where malicious \u0027speculative store bypass\u0027 can be used\n    and sanitize such patterns.\n\n     39: (bf) r3 \u003d r10\n     40: (07) r3 +\u003d -216\n     41: (79) r8 \u003d *(u64 *)(r7 +0)   // slow read\n     42: (7a) *(u64 *)(r10 -72) \u003d 0  // verifier inserts this instruction\n     43: (7b) *(u64 *)(r8 +0) \u003d r3   // this store becomes slow due to r8\n     44: (79) r1 \u003d *(u64 *)(r6 +0)   // cpu speculatively executes this load\n     45: (71) r2 \u003d *(u8 *)(r1 +0)    // speculatively arbitrary \u0027load byte\u0027\n                                     // is now sanitized\n\n    Above code after x86 JIT becomes:\n     e5: mov    %rbp,%rdx\n     e8: add    $0xffffffffffffff28,%rdx\n     ef: mov    0x0(%r13),%r14\n     f3: movq   $0x0,-0x48(%rbp)\n     fb: mov    %rdx,0x0(%r14)\n     ff: mov    0x0(%rbx),%rdi\n    103: movzbq 0x0(%rdi),%rsi\n\n    Signed-off-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Signed-off-by: Thomas Gleixner \u003ctglx@linutronix.de\u003e\n    [bwh: Backported to 4.9:\n     - Add bpf_verifier_env parameter to check_stack_write()\n     - Look up stack slot_types with state-\u003estack_slot_type[] rather than\n       state-\u003estack[].slot_type[]\n     - Drop bpf_verifier_env argument to verbose()\n     - Adjust context]\n    Signed-off-by: Ben Hutchings \u003cben.hutchings@codethink.co.uk\u003e\n    Signed-off-by: Greg Kroah-Hartman \u003cgregkh@linuxfoundation.org\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit d0bcf4dfce7a37a5895abaa2077ac86f515354f4\nAuthor: Ben Hutchings \u003cben.hutchings@codethink.co.uk\u003e\nDate:   Wed Dec 5 22:41:36 2018 +0000\n\n    bpf/verifier: Pass instruction index to check_mem_access() and check_xadd()\n\n    Extracted from commit 31fd85816dbe \"bpf: permits narrower load from\n    bpf program context fields\".\n\n    Cc: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Cc: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Signed-off-by: Ben Hutchings \u003cben.hutchings@codethink.co.uk\u003e\n    Signed-off-by: Greg Kroah-Hartman \u003cgregkh@linuxfoundation.org\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit c72922ee3cad14162e007be816f41500283ad353\nAuthor: Ben Hutchings \u003cben.hutchings@codethink.co.uk\u003e\nDate:   Wed Dec 5 22:45:15 2018 +0000\n\n    bpf/verifier: Add spi variable to check_stack_write()\n\n    Extracted from commit dc503a8ad984 \"bpf/verifier: track liveness for\n    pruning\".\n\n    Cc: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Cc: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Signed-off-by: Ben Hutchings \u003cben.hutchings@codethink.co.uk\u003e\n    Signed-off-by: Greg Kroah-Hartman \u003cgregkh@linuxfoundation.org\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 845cc9f56c69abd319303897fc6b2308724e986a\nAuthor: Jakub Kicinski \u003cjakub.kicinski@netronome.com\u003e\nDate:   Thu May 3 18:37:17 2018 -0700\n\n    bpf: fix references to free_bpf_prog_info() in comments\n\n    [ Upstream commit ab7f5bf0928be2f148d000a6eaa6c0a36e74750e ]\n\n    Comments in the verifier refer to free_bpf_prog_info() which\n    seems to have never existed in tree.  Replace it with\n    free_used_maps().\n\n    Signed-off-by: Jakub Kicinski \u003cjakub.kicinski@netronome.com\u003e\n    Reviewed-by: Quentin Monnet \u003cquentin.monnet@netronome.com\u003e\n    Signed-off-by: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Signed-off-by: Sasha Levin \u003calexander.levin@microsoft.com\u003e\n    Signed-off-by: Greg Kroah-Hartman \u003cgregkh@linuxfoundation.org\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit d75c2eef9a9ea46331dffcd9a3ffcd5aa0c688e8\nAuthor: Teng Qin \u003cqinteng@fb.com\u003e\nDate:   Mon Apr 24 19:00:37 2017 -0700\n\n    bpf: map_get_next_key to return first key on NULL\n\n    commit 8fe45924387be6b5c1be59a7eb330790c61d5d10 upstream.\n\n    When iterating through a map, we need to find a key that does not exist\n    in the map so map_get_next_key will give us the first key of the map.\n    This often requires a lot of guessing in production systems.\n\n    This patch makes map_get_next_key return the first key when the key\n    pointer in the parameter is NULL.\n\n    Signed-off-by: Teng Qin \u003cqinteng@fb.com\u003e\n    Signed-off-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Acked-by: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Chenbo Feng \u003cfengc@google.com\u003e\n    Cc: Lorenzo Colitti \u003clorenzo@google.com\u003e\n    Signed-off-by: Greg Kroah-Hartman \u003cgregkh@linuxfoundation.org\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit be491e4e2db6d51a4697d45d27e55a072d4570ab\nAuthor: Chenbo Feng \u003cfengc@google.com\u003e\nDate:   Mon Mar 19 17:57:27 2018 -0700\n\n    bpf: skip unnecessary capability check\n\n    commit 0fa4fe85f4724fff89b09741c437cbee9cf8b008 upstream.\n\n    The current check statement in BPF syscall will do a capability check\n    for CAP_SYS_ADMIN before checking sysctl_unprivileged_bpf_disabled. This\n    code path will trigger unnecessary security hooks on capability checking\n    and cause false alarms on unprivileged process trying to get CAP_SYS_ADMIN\n    access. This can be resolved by simply switch the order of the statement\n    and CAP_SYS_ADMIN is not required anyway if unprivileged bpf syscall is\n    allowed.\n\n    Signed-off-by: Chenbo Feng \u003cfengc@google.com\u003e\n    Acked-by: Lorenzo Colitti \u003clorenzo@google.com\u003e\n    Signed-off-by: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Signed-off-by: Greg Kroah-Hartman \u003cgregkh@linuxfoundation.org\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 0e5e2d5398ae4d88ecbc9e8160aa611bf57a1144\nAuthor: Al Viro \u003cviro@zeniv.linux.org.uk\u003e\nDate:   Sat Dec 2 20:20:38 2017 -0500\n\n    BACKPORT: fix \"netfilter: xt_bpf: Fix XT_BPF_MODE_FD_PINNED mode of \u0027xt_bpf_info_v1\u0027\"\n\n    Descriptor table is a shared object; it\u0027s not a place where you can\n    stick temporary references to files, especially when we don\u0027t need\n    an opened file at all.\n\n    Cc: stable@vger.kernel.org # v4.14\n    Fixes: 98589a0998b8 (\"netfilter: xt_bpf: Fix XT_BPF_MODE_FD_PINNED mode of \u0027xt_bpf_info_v1\u0027\")\n    Signed-off-by: Al Viro \u003cviro@zeniv.linux.org.uk\u003e\n    Signed-off-by: Chenbo Feng \u003cfengc@google.com\u003e\n\n    Removed the code related to function bpf_prog_get_ok() since it is not\n    exsit in current android tree.\n    (cherry picked from commit 040ee69226f8a96b7943645d68f41d5d44b5ff7d)\n\n    Change-Id: If7a602128cdea4b4b50c8effb215c9bca7449515\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 627ca6879b1afbdcd0e86116ade2a393784a881c\nAuthor: Jann Horn \u003cjannh@google.com\u003e\nDate:   Fri Dec 1 01:46:07 2017 +0100\n\n    UPSTREAM: netfilter: xt_bpf: add overflow checks\n\n    Check whether inputs from userspace are too long (explicit length field too\n    big or string not null-terminated) to avoid out-of-bounds reads.\n\n    As far as I can tell, this can at worst lead to very limited kernel heap\n    memory disclosure or oopses.\n\n    This bug can be triggered by an unprivileged user even if the xt_bpf module\n    is not loaded: iptables is available in network namespaces, and the xt_bpf\n    module can be autoloaded.\n\n    Triggering the bug with a classic BPF filter with fake length 0x1000 causes\n    the following KASAN report:\n\n    \u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\n    BUG: KASAN: slab-out-of-bounds in bpf_prog_create+0x84/0xf0\n    Read of size 32768 at addr ffff8801eff2c494 by task test/4627\n\n    CPU: 0 PID: 4627 Comm: test Not tainted 4.15.0-rc1+ #1\n    [...]\n    Call Trace:\n     dump_stack+0x5c/0x85\n     print_address_description+0x6a/0x260\n     kasan_report+0x254/0x370\n     ? bpf_prog_create+0x84/0xf0\n     memcpy+0x1f/0x50\n     bpf_prog_create+0x84/0xf0\n     bpf_mt_check+0x90/0xd6 [xt_bpf]\n    [...]\n    Allocated by task 4627:\n     kasan_kmalloc+0xa0/0xd0\n     __kmalloc_node+0x47/0x60\n     xt_alloc_table_info+0x41/0x70 [x_tables]\n    [...]\n    The buggy address belongs to the object at ffff8801eff2c3c0\n                    which belongs to the cache kmalloc-2048 of size 2048\n    The buggy address is located 212 bytes inside of\n                    2048-byte region [ffff8801eff2c3c0, ffff8801eff2cbc0)\n    [...]\n    \u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\u003d\n\n    Fixes: e6f30c731718 (\"netfilter: x_tables: add xt_bpf match\")\n    Signed-off-by: Jann Horn \u003cjannh@google.com\u003e\n    Signed-off-by: Pablo Neira Ayuso \u003cpablo@netfilter.org\u003e\n    Signed-off-by: Chenbo Feng \u003cfengc@google.com\u003e\n\n    (cherry picked from commit 6ab405114b0b229151ef06f4e31c7834dd09d0c0)\n\n    Change-Id: Ie066a9df84812853a9c9d2e51aa53646f4001542\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 9ad83cea953515af119c094eefd5cf15c7af288b\nAuthor: Shmulik Ladkani \u003cshmulik.ladkani@gmail.com\u003e\nDate:   Mon Oct 9 15:27:15 2017 +0300\n\n    UPSTREAM: netfilter: xt_bpf: Fix XT_BPF_MODE_FD_PINNED mode of \u0027xt_bpf_info_v1\u0027\n\n    Commit 2c16d6033264 (\"netfilter: xt_bpf: support ebpf\") introduced\n    support for attaching an eBPF object by an fd, with the\n    \u0027bpf_mt_check_v1\u0027 ABI expecting the \u0027.fd\u0027 to be specified upon each\n    IPT_SO_SET_REPLACE call.\n\n    However this breaks subsequent iptables calls:\n\n     # iptables -A INPUT -m bpf --object-pinned /sys/fs/bpf/xxx -j ACCEPT\n     # iptables -A INPUT -s 5.6.7.8 -j ACCEPT\n     iptables: Invalid argument. Run `dmesg\u0027 for more information.\n\n    That\u0027s because iptables works by loading existing rules using\n    IPT_SO_GET_ENTRIES to userspace, then issuing IPT_SO_SET_REPLACE with\n    the replacement set.\n\n    However, the loaded \u0027xt_bpf_info_v1\u0027 has an arbitrary \u0027.fd\u0027 number\n    (from the initial \"iptables -m bpf\" invocation) - so when 2nd invocation\n    occurs, userspace passes a bogus fd number, which leads to\n    \u0027bpf_mt_check_v1\u0027 to fail.\n\n    One suggested solution [1] was to hack iptables userspace, to perform a\n    \"entries fixup\" immediatley after IPT_SO_GET_ENTRIES, by opening a new,\n    process-local fd per every \u0027xt_bpf_info_v1\u0027 entry seen.\n\n    However, in [2] both Pablo Neira Ayuso and Willem de Bruijn suggested to\n    depricate the xt_bpf_info_v1 ABI dealing with pinned ebpf objects.\n\n    This fix changes the XT_BPF_MODE_FD_PINNED behavior to ignore the given\n    \u0027.fd\u0027 and instead perform an in-kernel lookup for the bpf object given\n    the provided \u0027.path\u0027.\n\n    It also defines an alias for the XT_BPF_MODE_FD_PINNED mode, named\n    XT_BPF_MODE_PATH_PINNED, to better reflect the fact that the user is\n    expected to provide the path of the pinned object.\n\n    Existing XT_BPF_MODE_FD_ELF behavior (non-pinned fd mode) is preserved.\n\n    References: [1] https://marc.info/?l\u003dnetfilter-devel\u0026m\u003d150564724607440\u0026w\u003d2\n                [2] https://marc.info/?l\u003dnetfilter-devel\u0026m\u003d150575727129880\u0026w\u003d2\n\n    Reported-by: Rafael Buchbinder \u003crafi@rbk.ms\u003e\n    Signed-off-by: Shmulik Ladkani \u003cshmulik.ladkani@gmail.com\u003e\n    Acked-by: Willem de Bruijn \u003cwillemb@google.com\u003e\n    Acked-by: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Signed-off-by: Pablo Neira Ayuso \u003cpablo@netfilter.org\u003e\n    Signed-off-by: Chenbo Feng \u003cfengc@google.com\u003e\n    (cherry picked from commit 98589a0998b8b13c4a8fa1ccb0e62751a019faa5)\n\n    Change-Id: Ia0d15a76823cca3afb38786a3d2c25c13ccf941d\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit b638511fb313784e17c3914b1974d7ff87a63a26\nAuthor: Willem de Bruijn \u003cwillemb@google.com\u003e\nDate:   Tue Dec 6 16:25:02 2016 -0500\n\n    UPSTREAM: netfilter: xt_bpf: support ebpf\n\n    Add support for attaching an eBPF object by file descriptor.\n\n    The iptables binary can be called with a path to an elf object or a\n    pinned bpf object. Also pass the mode and path to the kernel to be\n    able to return it later for iptables dump and save.\n\n    Signed-off-by: Willem de Bruijn \u003cwillemb@google.com\u003e\n    Signed-off-by: Pablo Neira Ayuso \u003cpablo@netfilter.org\u003e\n    Signed-off-by: Chenbo Feng \u003cfengc@google.com\u003e\n    (cherry picked from commit 2c16d60332643e90d4fa244f4a706c454b8c7569)\n\n    Change-Id: I31b8831a7ffd7c44985ee906ff194c1d934dafbe\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 940b385faa30f393a6041f39433b0b2a68cc86a3\nAuthor: Alexei Starovoitov \u003cast@fb.com\u003e\nDate:   Thu Sep 1 18:37:24 2016 -0700\n\n    perf, bpf: add perf events core support for BPF_PROG_TYPE_PERF_EVENT programs\n\n    Allow attaching BPF_PROG_TYPE_PERF_EVENT programs to sw and hw perf events\n    via overflow_handler mechanism.\n    When program is attached the overflow_handlers become stacked.\n    The program acts as a filter.\n    Returning zero from the program means that the normal perf_event_output handler\n    will not be called and sampling event won\u0027t be stored in the ring buffer.\n\n    The overflow_handler_context\u003d\u003dNULL is an additional safety check\n    to make sure programs are not attached to hw breakpoints and watchdog\n    in case other checks (that prevent that now anyway) get accidentally\n    relaxed in the future.\n\n    The program refcnt is incremented in case perf_events are inhereted\n    when target task is forked.\n    Similar to kprobe and tracepoint programs there is no ioctl to\n    detach the program or swap already attached program. The user space\n    expected to close(perf_event_fd) like it does right now for kprobe+bpf.\n    That restriction simplifies the code quite a bit.\n\n    The invocation of overflow_handler in __perf_event_overflow() is now\n    done via READ_ONCE, since that pointer can be replaced when the program\n    is attached while perf_event itself could have been active already.\n    There is no need to do similar treatment for event-\u003eprog, since it\u0027s\n    assigned only once before it\u0027s accessed.\n\n    Signed-off-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit cc8491eb8f8bcbebf6ef9b1fc284d81671f30194\nAuthor: Wang Nan \u003cwangnan0@huawei.com\u003e\nDate:   Mon Mar 28 06:41:30 2016 +0000\n\n    perf/core: Set event\u0027s default ::overflow_handler()\n\n    Set a default event-\u003eoverflow_handler in perf_event_alloc() so don\u0027t\n    need to check event-\u003eoverflow_handler in __perf_event_overflow().\n    Following commits can give a different default overflow_handler.\n\n    Initial idea comes from Peter:\n\n      http://lkml.kernel.org/r/20130708121557.GA17211@twins.programming.kicks-ass.net\n\n    Since the default value of event-\u003eoverflow_handler is not NULL, existing\n    \u0027if (!overflow_handler)\u0027 checks need to be changed.\n\n    is_default_overflow_handler() is introduced for this.\n\n    No extra performance overhead is introduced into the hot path because in the\n    original code we still need to read this handler from memory. A conditional\n    branch is avoided so actually we remove some instructions.\n\n    Signed-off-by: Wang Nan \u003cwangnan0@huawei.com\u003e\n    Signed-off-by: Peter Zijlstra (Intel) \u003cpeterz@infradead.org\u003e\n    Cc: \u003cpi3orama@163.com\u003e\n    Cc: Alexander Shishkin \u003calexander.shishkin@linux.intel.com\u003e\n    Cc: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Cc: Arnaldo Carvalho de Melo \u003cacme@redhat.com\u003e\n    Cc: Brendan Gregg \u003cbrendan.d.gregg@gmail.com\u003e\n    Cc: He Kuang \u003chekuang@huawei.com\u003e\n    Cc: Jiri Olsa \u003cjolsa@kernel.org\u003e\n    Cc: Jiri Olsa \u003cjolsa@redhat.com\u003e\n    Cc: Linus Torvalds \u003ctorvalds@linux-foundation.org\u003e\n    Cc: Masami Hiramatsu \u003cmasami.hiramatsu.pt@hitachi.com\u003e\n    Cc: Namhyung Kim \u003cnamhyung@kernel.org\u003e\n    Cc: Peter Zijlstra \u003cpeterz@infradead.org\u003e\n    Cc: Stephane Eranian \u003ceranian@google.com\u003e\n    Cc: Thomas Gleixner \u003ctglx@linutronix.de\u003e\n    Cc: Vince Weaver \u003cvincent.weaver@maine.edu\u003e\n    Cc: Zefan Li \u003clizefan@huawei.com\u003e\n    Link: http://lkml.kernel.org/r/1459147292-239310-3-git-send-email-wangnan0@huawei.com\n    Signed-off-by: Ingo Molnar \u003cmingo@kernel.org\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 80032c957d6ed2428d5a79fc43191904bf615683\nAuthor: Eric Dumazet \u003cedumazet@google.com\u003e\nDate:   Thu Mar 8 16:17:36 2018 +0100\n\n    bpf: add schedule points in percpu arrays management\n\n    [ upstream commit 32fff239de37ef226d5b66329dd133f64d63b22d ]\n\n    syszbot managed to trigger RCU detected stalls in\n    bpf_array_free_percpu()\n\n    It takes time to allocate a huge percpu map, but even more time to free\n    it.\n\n    Since we run in process context, use cond_resched() to yield cpu if\n    needed.\n\n    Fixes: a10423b87a7e (\"bpf: introduce BPF_MAP_TYPE_PERCPU_ARRAY map\")\n    Signed-off-by: Eric Dumazet \u003cedumazet@google.com\u003e\n    Reported-by: syzbot \u003csyzkaller@googlegroups.com\u003e\n    Signed-off-by: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Signed-off-by: Greg Kroah-Hartman \u003cgregkh@linuxfoundation.org\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit c482cc7b23ae0090676a70e1a302d6948b904c87\nAuthor: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\nDate:   Thu Mar 8 16:17:33 2018 +0100\n\n    bpf: fix mlock precharge on arraymaps\n\n    [ upstream commit 9c2d63b843a5c8a8d0559cc067b5398aa5ec3ffc ]\n\n    syzkaller recently triggered OOM during percpu map allocation;\n    while there is work in progress by Dennis Zhou to add __GFP_NORETRY\n    semantics for percpu allocator under pressure, there seems also a\n    missing bpf_map_precharge_memlock() check in array map allocation.\n\n    Given today the actual bpf_map_charge_memlock() happens after the\n    find_and_alloc_map() in syscall path, the bpf_map_precharge_memlock()\n    is there to bail out early before we go and do the map setup work\n    when we find that we hit the limits anyway. Therefore add this for\n    array map as well.\n\n    Fixes: 6c9059817432 (\"bpf: pre-allocate hash map elements\")\n    Fixes: a10423b87a7e (\"bpf: introduce BPF_MAP_TYPE_PERCPU_ARRAY map\")\n    Reported-by: syzbot+adb03f3f0bb57ce3acda@syzkaller.appspotmail.com\n    Signed-off-by: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Cc: Dennis Zhou \u003cdennisszhou@gmail.com\u003e\n    Signed-off-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Signed-off-by: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Signed-off-by: Greg Kroah-Hartman \u003cgregkh@linuxfoundation.org\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit b7e8017c7cbff87a6c58f3e7babb304b78b7f6ff\nAuthor: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\nDate:   Thu Mar 8 16:17:32 2018 +0100\n\n    bpf: fix wrong exposure of map_flags into fdinfo for lpm\n\n    [ upstream commit a316338cb71a3260201490e615f2f6d5c0d8fb2c ]\n\n    trie_alloc() always needs to have BPF_F_NO_PREALLOC passed in via\n    attr-\u003emap_flags, since it does not support preallocation yet. We\n    check the flag, but we never copy the flag into trie-\u003emap.map_flags,\n    which is later on exposed into fdinfo and used by loaders such as\n    iproute2. Latter uses this in bpf_map_selfcheck_pinned() to test\n    whether a pinned map has the same spec as the one from the BPF obj\n    file and if not, bails out, which is currently the case for lpm\n    since it exposes always 0 as flags.\n\n    Also copy over flags in array_map_alloc() and stack_map_alloc().\n    They always have to be 0 right now, but we should make sure to not\n    miss to copy them over at a later point in time when we add actual\n    flags for them to use.\n\n    Fixes: b95a5c4db09b (\"bpf: add a longest prefix match trie map implementation\")\n    Reported-by: Jarno Rajahalme \u003cjarno@covalent.io\u003e\n    Signed-off-by: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Acked-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Signed-off-by: Greg Kroah-Hartman \u003cgregkh@linuxfoundation.org\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit d2ec721043f8adefb4f1bd9f281cc15cdbe834be\nAuthor: Sami Tolvanen \u003csamitolvanen@google.com\u003e\nDate:   Thu Aug 24 08:59:31 2017 -0700\n\n    bpf: fix function type for __bpf_prog_run\n\n    Bug: 67506682\n    Change-Id: I096a470c65a2a1867c51da9a33843ae23bf5e547\n    Signed-off-by: Sami Tolvanen \u003csamitolvanen@google.com\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 8bd8794ffdd831dd10e19d438f93feda0ecef542\nAuthor: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\nDate:   Mon Jan 29 02:49:01 2018 +0100\n\n    bpf: reject stores into ctx via st and xadd\n\n    [ upstream commit f37a8cb84cce18762e8f86a70bd6a49a66ab964c ]\n\n    Alexei found that verifier does not reject stores into context\n    via BPF_ST instead of BPF_STX. And while looking at it, we\n    also should not allow XADD variant of BPF_STX.\n\n    The context rewriter is only assuming either BPF_LDX_MEM- or\n    BPF_STX_MEM-type operations, thus reject anything other than\n    that so that assumptions in the rewriter properly hold. Add\n    test cases as well for BPF selftests.\n\n    Fixes: d691f9e8d440 (\"bpf: allow programs to write to certain skb fields\")\n    Reported-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Signed-off-by: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Signed-off-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Signed-off-by: Greg Kroah-Hartman \u003cgregkh@linuxfoundation.org\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 2bd72a6c616db15e1a8f5383815848c9501b8909\nAuthor: Alexei Starovoitov \u003cast@kernel.org\u003e\nDate:   Mon Jan 29 02:49:00 2018 +0100\n\n    bpf: fix 32-bit divide by zero\n\n    [ upstream commit 68fda450a7df51cff9e5a4d4a4d9d0d5f2589153 ]\n\n    due to some JITs doing if (src_reg \u003d\u003d 0) check in 64-bit mode\n    for div/mod operations mask upper 32-bits of src register\n    before doing the check\n\n    Fixes: 622582786c9e (\"net: filter: x86: internal BPF JIT\")\n    Fixes: 7a12b5031c6b (\"sparc64: Add eBPF JIT.\")\n    Reported-by: syzbot+48340bb518e88849e2e3@syzkaller.appspotmail.com\n    Signed-off-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Signed-off-by: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Signed-off-by: Greg Kroah-Hartman \u003cgregkh@linuxfoundation.org\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit e164dcc1b8be97d6a921f97729a1988e348ae1bb\nAuthor: Eric Dumazet \u003cedumazet@google.com\u003e\nDate:   Mon Jan 29 02:48:59 2018 +0100\n\n    bpf: fix divides by zero\n\n    [ upstream commit c366287ebd698ef5e3de300d90cd62ee9ee7373e ]\n\n    Divides by zero are not nice, lets avoid them if possible.\n\n    Also do_div() seems not needed when dealing with 32bit operands,\n    but this seems a minor detail.\n\n    Fixes: bd4cf0ed331a (\"net: filter: rework/optimize internal BPF interpreter\u0027s instruction set\")\n    Signed-off-by: Eric Dumazet \u003cedumazet@google.com\u003e\n    Reported-by: syzbot \u003csyzkaller@googlegroups.com\u003e\n    Signed-off-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Signed-off-by: Greg Kroah-Hartman \u003cgregkh@linuxfoundation.org\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 657456a8c557c2dbe4605a488c8280ad97192393\nAuthor: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\nDate:   Mon Jan 29 02:48:57 2018 +0100\n\n    bpf: arsh is not supported in 32 bit alu thus reject it\n\n    [ upstream commit 7891a87efc7116590eaba57acc3c422487802c6f ]\n\n    The following snippet was throwing an \u0027unknown opcode cc\u0027 warning\n    in BPF interpreter:\n\n      0: (18) r0 \u003d 0x0\n      2: (7b) *(u64 *)(r10 -16) \u003d r0\n      3: (cc) (u32) r0 s\u003e\u003e\u003d (u32) r0\n      4: (95) exit\n\n    Although a number of JITs do support BPF_ALU | BPF_ARSH | BPF_{K,X}\n    generation, not all of them do and interpreter does neither. We can\n    leave existing ones and implement it later in bpf-next for the\n    remaining ones, but reject this properly in verifier for the time\n    being.\n\n    Fixes: 17a5267067f3 (\"bpf: verifier (add verifier core)\")\n    Reported-by: syzbot+93c4904c5c70348a6890@syzkaller.appspotmail.com\n    Signed-off-by: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Signed-off-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Signed-off-by: Greg Kroah-Hartman \u003cgregkh@linuxfoundation.org\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 0a62818fdea2324b7d2a9b6cd718dfc03dc2364e\nAuthor: Alexei Starovoitov \u003cast@kernel.org\u003e\nDate:   Mon Jan 29 02:48:56 2018 +0100\n\n    bpf: introduce BPF_JIT_ALWAYS_ON config\n\n    [ upstream commit 290af86629b25ffd1ed6232c4e9107da031705cb ]\n\n    The BPF interpreter has been used as part of the spectre 2 attack CVE-2017-5715.\n\n    A quote from goolge project zero blog:\n    \"At this point, it would normally be necessary to locate gadgets in\n    the host kernel code that can be used to actually leak data by reading\n    from an attacker-controlled location, shifting and masking the result\n    appropriately and then using the result of that as offset to an\n    attacker-controlled address for a load. But piecing gadgets together\n    and figuring out which ones work in a speculation context seems annoying.\n    So instead, we decided to use the eBPF interpreter, which is built into\n    the host kernel - while there is no legitimate way to invoke it from inside\n    a VM, the presence of the code in the host kernel\u0027s text section is sufficient\n    to make it usable for the attack, just like with ordinary ROP gadgets.\"\n\n    To make attacker job harder introduce BPF_JIT_ALWAYS_ON config\n    option that removes interpreter from the kernel in favor of JIT-only mode.\n    So far eBPF JIT is supported by:\n    x64, arm64, arm32, sparc64, s390, powerpc64, mips64\n\n    The start of JITed program is randomized and code page is marked as read-only.\n    In addition \"constant blinding\" can be turned on with net.core.bpf_jit_harden\n\n    v2-\u003ev3:\n    - move __bpf_prog_ret0 under ifdef (Daniel)\n\n    v1-\u003ev2:\n    - fix init order, test_bpf and cBPF (Daniel\u0027s feedback)\n    - fix offloaded bpf (Jakub\u0027s feedback)\n    - add \u0027return 0\u0027 dummy in case something can invoke prog-\u003ebpf_func\n    - retarget bpf tree. For bpf-next the patch would need one extra hunk.\n      It will be sent when the trees are merged back to net-next\n\n    Considered doing:\n      int bpf_jit_enable __read_mostly \u003d BPF_EBPF_JIT_DEFAULT;\n    but it seems better to land the patch as-is and in bpf-next remove\n    bpf_jit_enable global variable from all JITs, consolidate in one place\n    and remove this jit_init() function.\n\n    Signed-off-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Signed-off-by: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Signed-off-by: Greg Kroah-Hartman \u003cgregkh@linuxfoundation.org\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 3f86a40ef67faa3933d265b54ab990f55d64d369\nAuthor: Alexei Starovoitov \u003cast@fb.com\u003e\nDate:   Mon Jan 29 02:48:55 2018 +0100\n\n    bpf: fix bpf_tail_call() x64 JIT\n\n    [ upstream commit 90caccdd8cc0215705f18b92771b449b01e2474a ]\n\n    - bpf prog_array just like all other types of bpf array accepts 32-bit index.\n      Clarify that in the comment.\n    - fix x64 JIT of bpf_tail_call which was incorrectly loading 8 instead of 4 bytes\n    - tighten corresponding check in the interpreter to stay consistent\n\n    The JIT bug can be triggered after introduction of BPF_F_NUMA_NODE flag\n    in commit 96eabe7a40aa in 4.14. Before that the map_flags would stay zero and\n    though JIT code is wrong it will check bounds correctly.\n    Hence two fixes tags. All other JITs don\u0027t have this problem.\n\n    Signed-off-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Fixes: 96eabe7a40aa (\"bpf: Allow selecting numa node during map creation\")\n    Fixes: b52f00e6a715 (\"x86: bpf_jit: implement bpf_tail_call() helper\")\n    Acked-by: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Acked-by: Martin KaFai Lau \u003ckafai@fb.com\u003e\n    Reviewed-by: Eric Dumazet \u003cedumazet@google.com\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Greg Kroah-Hartman \u003cgregkh@linuxfoundation.org\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit f97771fcebf3fdd241a34838519c38d459452c0c\nAuthor: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\nDate:   Wed Jan 10 23:25:05 2018 +0100\n\n    bpf, array: fix overflow in max_entries and undefined behavior in index_mask\n\n    commit bbeb6e4323dad9b5e0ee9f60c223dd532e2403b1 upstream.\n\n    syzkaller tried to alloc a map with 0xfffffffd entries out of a userns,\n    and thus unprivileged. With the recently added logic in b2157399cc98\n    (\"bpf: prevent out-of-bounds speculation\") we round this up to the next\n    power of two value for max_entries for unprivileged such that we can\n    apply proper masking into potentially zeroed out map slots.\n\n    However, this will generate an index_mask of 0xffffffff, and therefore\n    a + 1 will let this overflow into new max_entries of 0. This will pass\n    allocation, etc, and later on map access we still enforce on the original\n    attr-\u003emax_entries value which was 0xfffffffd, therefore triggering GPF\n    all over the place. Thus bail out on overflow in such case.\n\n    Moreover, on 32 bit archs roundup_pow_of_two() can also not be used,\n    since fls_long(max_entries - 1) can result in 32 and 1UL \u003c\u003c 32 in 32 bit\n    space is undefined. Therefore, do this by hand in a 64 bit variable.\n\n    This fixes all the issues triggered by syzkaller\u0027s reproducers.\n\n    Fixes: b2157399cc98 (\"bpf: prevent out-of-bounds speculation\")\n    Reported-by: syzbot+b0efb8e572d01bce1ae0@syzkaller.appspotmail.com\n    Reported-by: syzbot+6c15e9744f75f2364773@syzkaller.appspotmail.com\n    Reported-by: syzbot+d2f5524fb46fd3b312ee@syzkaller.appspotmail.com\n    Reported-by: syzbot+61d23c95395cc90dbc2b@syzkaller.appspotmail.com\n    Reported-by: syzbot+0d363c942452cca68c01@syzkaller.appspotmail.com\n    Signed-off-by: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Signed-off-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Signed-off-by: Greg Kroah-Hartman \u003cgregkh@linuxfoundation.org\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit c978b96da9b50b634930fca5e1f20dc8f480525a\nAuthor: Alexei Starovoitov \u003cast@kernel.org\u003e\nDate:   Sun Jan 7 17:33:02 2018 -0800\n\n    bpf: prevent out-of-bounds speculation\n\n    commit b2157399cc9898260d6031c5bfe45fe137c1fbe7 upstream.\n\n    Under speculation, CPUs may mis-predict branches in bounds checks. Thus,\n    memory accesses under a bounds check may be speculated even if the\n    bounds check fails, providing a primitive for building a side channel.\n\n    To avoid leaking kernel data round up array-based maps and mask the index\n    after bounds check, so speculated load with out of bounds index will load\n    either valid value from the array or zero from the padded area.\n\n    Unconditionally mask index for all array types even when max_entries\n    are not rounded to power of 2 for root user.\n    When map is created by unpriv user generate a sequence of bpf insns\n    that includes AND operation to make sure that JITed code includes\n    the same \u0027index \u0026 index_mask\u0027 operation.\n\n    If prog_array map is created by unpriv user replace\n      bpf_tail_call(ctx, map, index);\n    with\n      if (index \u003e\u003d max_entries) {\n        index \u0026\u003d map-\u003eindex_mask;\n        bpf_tail_call(ctx, map, index);\n      }\n    (along with roundup to power 2) to prevent out-of-bounds speculation.\n    There is secondary redundant \u0027if (index \u003e\u003d max_entries)\u0027 in the interpreter\n    and in all JITs, but they can be optimized later if necessary.\n\n    Other array-like maps (cpumap, devmap, sockmap, perf_event_array, cgroup_array)\n    cannot be used by unpriv, so no changes there.\n\n    That fixes bpf side of \"Variant 1: bounds check bypass (CVE-2017-5753)\" on\n    all architectures with and without JIT.\n\n    v2-\u003ev3:\n    Daniel noticed that attack potentially can be crafted via syscall commands\n    without loading the program, so add masking to those paths as well.\n\n    Signed-off-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Acked-by: John Fastabend \u003cjohn.fastabend@gmail.com\u003e\n    Signed-off-by: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Cc: Jiri Slaby \u003cjslaby@suse.cz\u003e\n    [ Backported to 4.9 - gregkh ]\n    Signed-off-by: Greg Kroah-Hartman \u003cgregkh@linuxfoundation.org\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit de3a6d32e3ac9201e4af128c5f70381465543173\nAuthor: Alexei Starovoitov \u003cast@fb.com\u003e\nDate:   Wed Mar 15 18:26:40 2017 -0700\n\n    bpf: refactor fixup_bpf_calls()\n\n    commit 79741b3bdec01a8628368fbcfccc7d189ed606cb upstream.\n\n    reduce indent and make it iterate over instructions similar to\n    convert_ctx_accesses(). Also convert hard BUG_ON into soft verifier error.\n\n    Signed-off-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Acked-by: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Cc: Jiri Slaby \u003cjslaby@suse.cz\u003e\n    [Backported to 4.9.y - gregkh]\n    Signed-off-by: Greg Kroah-Hartman \u003cgregkh@linuxfoundation.org\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 269e33f4dbd8843baea95b5515d5394d403e8e34\nAuthor: Alexei Starovoitov \u003cast@fb.com\u003e\nDate:   Wed Mar 15 18:26:39 2017 -0700\n\n    bpf: move fixup_bpf_calls() function\n\n    commit e245c5c6a5656e4d61aa7bb08e9694fd6e5b2b9d upstream.\n\n    no functional change.\n    move fixup_bpf_calls() to verifier.c\n    it\u0027s being refactored in the next patch\n\n    Signed-off-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Acked-by: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Cc: Jiri Slaby \u003cjslaby@suse.cz\u003e\n    [backported to 4.9 - gregkh]\n    Signed-off-by: Greg Kroah-Hartman \u003cgregkh@linuxfoundation.org\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 5ea68e2ac361d6ebc428c6bcad900606b62aaecb\nAuthor: Ben Hutchings \u003cben@decadent.org.uk\u003e\nDate:   Sat Dec 23 02:26:17 2017 +0000\n\n    bpf/verifier: Fix states_equal() comparison of pointer and UNKNOWN\n\n    An UNKNOWN_VALUE is not supposed to be derived from a pointer, unless\n    pointer leaks are allowed.  Therefore, states_equal() must not treat\n    a state with a pointer in a register as \"equal\" to a state with an\n    UNKNOWN_VALUE in that register.\n\n    This was fixed differently upstream, but the code around here was\n    largely rewritten in 4.14 by commit f1174f77b50c \"bpf/verifier: rework\n    value tracking\".  The bug can be detected by the bpf/verifier sub-test\n    \"pointer/scalar confusion in state equality check (way 1)\".\n\n    Signed-off-by: Ben Hutchings \u003cben@decadent.org.uk\u003e\n    Cc: Edward Cree \u003cecree@solarflare.com\u003e\n    Cc: Jann Horn \u003cjannh@google.com\u003e\n    Cc: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Cc: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit ad66e542393a362f979878d0ebc2c9b5f9a8539b\nAuthor: Jann Horn \u003cjannh@google.com\u003e\nDate:   Fri Dec 22 16:29:05 2017 +0100\n\n    bpf: fix incorrect sign extension in check_alu_op()\n\n    [ Upstream commit 95a762e2c8c942780948091f8f2a4f32fce1ac6f ]\n\n    Distinguish between\n    BPF_ALU64|BPF_MOV|BPF_K (load 32-bit immediate, sign-extended to 64-bit)\n    and BPF_ALU|BPF_MOV|BPF_K (load 32-bit immediate, zero-padded to 64-bit);\n    only perform sign extension in the first case.\n\n    Starting with v4.14, this is exploitable by unprivileged users as long as\n    the unprivileged_bpf_disabled sysctl isn\u0027t set.\n\n    Debian assigned CVE-2017-16995 for this issue.\n\n    v3:\n     - add CVE number (Ben Hutchings)\n\n    Fixes: 484611357c19 (\"bpf: allow access into map value arrays\")\n    Signed-off-by: Jann Horn \u003cjannh@google.com\u003e\n    Acked-by: Edward Cree \u003cecree@solarflare.com\u003e\n    Signed-off-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Signed-off-by: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Signed-off-by: Greg Kroah-Hartman \u003cgregkh@linuxfoundation.org\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 70dcc0306cfa4376f89bb3751bbebb91c12f1fd2\nAuthor: Jann Horn \u003cjannh@google.com\u003e\nDate:   Fri Dec 22 16:29:04 2017 +0100\n\n    bpf: reject out-of-bounds stack pointer calculation\n\n    Reject programs that compute wildly out-of-bounds stack pointers.\n    Otherwise, pointers can be computed with an offset that doesn\u0027t fit into an\n    `int`, causing security issues in the stack memory access check (as well as\n    signed integer overflow during offset addition).\n\n    This is a fix specifically for the v4.9 stable tree because the mainline\n    code looks very different at this point.\n\n    Fixes: 7bca0a9702edf (\"bpf: enhance verifier to understand stack pointer arithmetic\")\n    Signed-off-by: Jann Horn \u003cjannh@google.com\u003e\n    Acked-by: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Signed-off-by: Greg Kroah-Hartman \u003cgregkh@linuxfoundation.org\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 27cd18137a9bb4838880de77988ed9ccdc263ab3\nAuthor: Alexei Starovoitov \u003cast@fb.com\u003e\nDate:   Fri Dec 22 16:29:03 2017 +0100\n\n    bpf: fix branch pruning logic\n\n    [ Upstream commit c131187db2d3fa2f8bf32fdf4e9a4ef805168467 ]\n\n    when the verifier detects that register contains a runtime constant\n    and it\u0027s compared with another constant it will prune exploration\n    of the branch that is guaranteed not to be taken at runtime.\n    This is all correct, but malicious program may be constructed\n    in such a way that it always has a constant comparison and\n    the other branch is never taken under any conditions.\n    In this case such path through the program will not be explored\n    by the verifier. It won\u0027t be taken at run-time either, but since\n    all instructions are JITed the malicious program may cause JITs\n    to complain about using reserved fields, etc.\n    To fix the issue we have to track the instructions explored by\n    the verifier and sanitize instructions that are dead at run time\n    with NOPs. We cannot reject such dead code, since llvm generates\n    it for valid C code, since it doesn\u0027t do as much data flow\n    analysis as the verifier does.\n\n    Fixes: 17a5267067f3 (\"bpf: verifier (add verifier core)\")\n    Signed-off-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Acked-by: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Signed-off-by: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Signed-off-by: Greg Kroah-Hartman \u003cgregkh@linuxfoundation.org\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 608af3eebb9a43897f9d3240dc6fa6b7a2f1359b\nAuthor: Alexei Starovoitov \u003cast@fb.com\u003e\nDate:   Fri Dec 22 16:29:02 2017 +0100\n\n    bpf: adjust insn_aux_data when patching insns\n\n    [ Upstream commit 8041902dae5299c1f194ba42d14383f734631009 ]\n\n    convert_ctx_accesses() replaces single bpf instruction with a set of\n    instructions. Adjust corresponding insn_aux_data while patching.\n    It\u0027s needed to make sure subsequent \u0027for(all insn)\u0027 loops\n    have matching insn and insn_aux_data.\n\n    Signed-off-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Acked-by: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Greg Kroah-Hartman \u003cgregkh@linuxfoundation.org\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 0e35213cc824850a9d03968acd647634346b0a1a\nAuthor: Eric Dumazet \u003cedumazet@google.com\u003e\nDate:   Tue Nov 14 17:15:50 2017 -0800\n\n    bpf: fix lockdep splat\n\n    [ Upstream commit 89ad2fa3f043a1e8daae193bcb5fe34d5f8caf28 ]\n\n    pcpu_freelist_pop() needs the same lockdep awareness than\n    pcpu_freelist_populate() to avoid a false positive.\n\n     [ INFO: SOFTIRQ-safe -\u003e SOFTIRQ-unsafe lock order detected ]\n\n     switchto-defaul/12508 [HC0[0]:SC0[6]:HE0:SE0] is trying to acquire:\n      (\u0026htab-\u003ebuckets[i].lock){......}, at: [\u003cffffffff9dc099cb\u003e] __htab_percpu_map_update_elem+0x1cb/0x300\n\n     and this task is already holding:\n      (dev_queue-\u003edev-\u003eqdisc_class ?: \u0026qdisc_tx_lock#2){+.-...}, at: [\u003cffffffff9e135848\u003e] __dev_queue_xmit+0\n    x868/0x1240\n     which would create a new lock dependency:\n      (dev_queue-\u003edev-\u003eqdisc_class ?: \u0026qdisc_tx_lock#2){+.-...} -\u003e (\u0026htab-\u003ebuckets[i].lock){......}\n\n     but this new dependency connects a SOFTIRQ-irq-safe lock:\n      (dev_queue-\u003edev-\u003eqdisc_class ?: \u0026qdisc_tx_lock#2){+.-...}\n     ... which became SOFTIRQ-irq-safe at:\n       [\u003cffffffff9db5931b\u003e] __lock_acquire+0x42b/0x1f10\n       [\u003cffffffff9db5b32c\u003e] lock_acquire+0xbc/0x1b0\n       [\u003cffffffff9da05e38\u003e] _raw_spin_lock+0x38/0x50\n       [\u003cffffffff9e135848\u003e] __dev_queue_xmit+0x868/0x1240\n       [\u003cffffffff9e136240\u003e] dev_queue_xmit+0x10/0x20\n       [\u003cffffffff9e1965d9\u003e] ip_finish_output2+0x439/0x590\n       [\u003cffffffff9e197410\u003e] ip_finish_output+0x150/0x2f0\n       [\u003cffffffff9e19886d\u003e] ip_output+0x7d/0x260\n       [\u003cffffffff9e19789e\u003e] ip_local_out+0x5e/0xe0\n       [\u003cffffffff9e197b25\u003e] ip_queue_xmit+0x205/0x620\n       [\u003cffffffff9e1b8398\u003e] tcp_transmit_skb+0x5a8/0xcb0\n       [\u003cffffffff9e1ba152\u003e] tcp_write_xmit+0x242/0x1070\n       [\u003cffffffff9e1baffc\u003e] __tcp_push_pending_frames+0x3c/0xf0\n       [\u003cffffffff9e1b3472\u003e] tcp_rcv_established+0x312/0x700\n       [\u003cffffffff9e1c1acc\u003e] tcp_v4_do_rcv+0x11c/0x200\n       [\u003cffffffff9e1c3dc2\u003e] tcp_v4_rcv+0xaa2/0xc30\n       [\u003cffffffff9e191107\u003e] ip_local_deliver_finish+0xa7/0x240\n       [\u003cffffffff9e191a36\u003e] ip_local_deliver+0x66/0x200\n       [\u003cffffffff9e19137d\u003e] ip_rcv_finish+0xdd/0x560\n       [\u003cffffffff9e191e65\u003e] ip_rcv+0x295/0x510\n       [\u003cffffffff9e12ff88\u003e] __netif_receive_skb_core+0x988/0x1020\n       [\u003cffffffff9e130641\u003e] __netif_receive_skb+0x21/0x70\n       [\u003cffffffff9e1306ff\u003e] process_backlog+0x6f/0x230\n       [\u003cffffffff9e132129\u003e] net_rx_action+0x229/0x420\n       [\u003cffffffff9da07ee8\u003e] __do_softirq+0xd8/0x43d\n       [\u003cffffffff9e282bcc\u003e] do_softirq_own_stack+0x1c/0x30\n       [\u003cffffffff9dafc2f5\u003e] do_softirq+0x55/0x60\n       [\u003cffffffff9dafc3a8\u003e] __local_bh_enable_ip+0xa8/0xb0\n       [\u003cffffffff9db4c727\u003e] cpu_startup_entry+0x1c7/0x500\n       [\u003cffffffff9daab333\u003e] start_secondary+0x113/0x140\n\n     to a SOFTIRQ-irq-unsafe lock:\n      (\u0026head-\u003elock){+.+...}\n     ... which became SOFTIRQ-irq-unsafe at:\n     ...  [\u003cffffffff9db5971f\u003e] __lock_acquire+0x82f/0x1f10\n       [\u003cffffffff9db5b32c\u003e] lock_acquire+0xbc/0x1b0\n       [\u003cffffffff9da05e38\u003e] _raw_spin_lock+0x38/0x50\n       [\u003cffffffff9dc0b7fa\u003e] pcpu_freelist_pop+0x7a/0xb0\n       [\u003cffffffff9dc08b2c\u003e] htab_map_alloc+0x50c/0x5f0\n       [\u003cffffffff9dc00dc5\u003e] SyS_bpf+0x265/0x1200\n       [\u003cffffffff9e28195f\u003e] entry_SYSCALL_64_fastpath+0x12/0x17\n\n     other info that might help us debug this:\n\n     Chain exists of:\n       dev_queue-\u003edev-\u003eqdisc_class ?: \u0026qdisc_tx_lock#2 --\u003e \u0026htab-\u003ebuckets[i].lock --\u003e \u0026head-\u003elock\n\n      Possible interrupt unsafe locking scenario:\n\n            CPU0                    CPU1\n            ----                    ----\n       lock(\u0026head-\u003elock);\n                                    local_irq_disable();\n                                    lock(dev_queue-\u003edev-\u003eqdisc_class ?: \u0026qdisc_tx_lock#2);\n                                    lock(\u0026htab-\u003ebuckets[i].lock);\n       \u003cInterrupt\u003e\n         lock(dev_queue-\u003edev-\u003eqdisc_class ?: \u0026qdisc_tx_lock#2);\n\n      *** DEADLOCK ***\n\n    Fixes: e19494edab82 (\"bpf: introduce percpu_freelist\")\n    Signed-off-by: Eric Dumazet \u003cedumazet@google.com\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Sasha Levin \u003calexander.levin@verizon.com\u003e\n    Signed-off-by: Greg Kroah-Hartman \u003cgregkh@linuxfoundation.org\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 68477ac5434b528166dcd143aea9f2a01689204c\nAuthor: Chenbo Feng \u003cfengc@google.com\u003e\nDate:   Wed Oct 18 13:00:26 2017 -0700\n\n    UPSTREAM: selinux: bpf: Add addtional check for bpf object file receive\n\n    Introduce a bpf object related check when sending and receiving files\n    through unix domain socket as well as binder. It checks if the receiving\n    process have privilege to read/write the bpf map or use the bpf program.\n    This check is necessary because the bpf maps and programs are using a\n    anonymous inode as their shared inode so the normal way of checking the\n    files and sockets when passing between processes cannot work properly on\n    eBPF object. This check only works when the BPF_SYSCALL is configured.\n\n    Signed-off-by: Chenbo Feng \u003cfengc@google.com\u003e\n    Acked-by: Stephen Smalley \u003csds@tycho.nsa.gov\u003e\n    Reviewed-by: James Morris \u003cjames.l.morris@oracle.com\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n\n    (cherry-pick from net-next: f66e448cfda021b0bcd884f26709796fe19c7cc1)\n    Bug: 30950746\n\n    Change-Id: I5b2cf4ccb4eab7eda91ddd7091d6aa3e7ed9f2cd\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit e4e8ff5dfd1140bc696704f1ecb07237d85d4491\nAuthor: Chenbo Feng \u003cfengc@google.com\u003e\nDate:   Wed Oct 18 13:00:25 2017 -0700\n\n    UPSTREAM: selinux: bpf: Add selinux check for eBPF syscall operations\n\n    Implement the actual checks introduced to eBPF related syscalls. This\n    implementation use the security field inside bpf object to store a sid that\n    identify the bpf object. And when processes try to access the object,\n    selinux will check if processes have the right privileges. The creation\n    of eBPF object are also checked at the general bpf check hook and new\n    cmd introduced to eBPF domain can also be checked there.\n\n    Signed-off-by: Chenbo Feng \u003cfengc@google.com\u003e\n    Acked-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Reviewed-by: James Morris \u003cjames.l.morris@oracle.com\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n\n    (cherry-pick from net-next: ec27c3568a34c7fe5fcf4ac0a354eda77687f7eb)\n    Bug: 30950746\n    Change-Id: Ifb0cdd4b7d470223b143646b339ba511ac77c156\n\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n    Change-Id: If49bc4c91c152efd36b372f8dfa15486e916f7df\n\ncommit 05a054c2f0674a447ba5a157886ea2097f295dcd\nAuthor: Chenbo Feng \u003cfengc@google.com\u003e\nDate:   Wed Oct 18 13:00:24 2017 -0700\n\n    BACKPORT: security: bpf: Add LSM hooks for bpf object related syscall\n\n    Introduce several LSM hooks for the syscalls that will allow the\n    userspace to access to eBPF object such as eBPF programs and eBPF maps.\n    The security check is aimed to enforce a per object security protection\n    for eBPF object so only processes with the right priviliges can\n    read/write to a specific map or use a specific eBPF program. Besides\n    that, a general security hook is added before the multiplexer of bpf\n    syscall to check the cmd and the attribute used for the command. The\n    actual security module can decide which command need to be checked and\n    how the cmd should be checked.\n\n    Signed-off-by: Chenbo Feng \u003cfengc@google.com\u003e\n    Acked-by: James Morris \u003cjames.l.morris@oracle.com\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n\n    Added the LIST_HEAD_INIT call for security hooks, it nolonger exist in\n    uptream code.\n    (cherry-pick from net-next: afdb09c720b62b8090584c11151d856df330e57d)\n    Bug: 30950746\n\n    Change-Id: Ieb3ac74392f531735fc7c949b83346a5f587a77b\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 408ae54c568b225134119116a05b3ed8ae90ec55\nAuthor: Chenbo Feng \u003cfengc@google.com\u003e\nDate:   Wed Oct 18 13:00:22 2017 -0700\n\n    BACKPORT: bpf: Add file mode configuration into bpf maps\n\n    Introduce the map read/write flags to the eBPF syscalls that returns the\n    map fd. The flags is used to set up the file mode when construct a new\n    file descriptor for bpf maps. To not break the backward capability, the\n    f_flags is set to O_RDWR if the flag passed by syscall is 0. Otherwise\n    it should be O_RDONLY or O_WRONLY. When the userspace want to modify or\n    read the map content, it will check the file mode to see if it is\n    allowed to make the change.\n\n    Signed-off-by: Chenbo Feng \u003cfengc@google.com\u003e\n    Acked-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Acked-by: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n\n    Deleted the file mode configuration code in unsupported map type and\n    removed the file mode check in non-existing helper functions.\n    (cherry-pick from net-next: 6e71b04a82248ccf13a94b85cbc674a9fefe53f5)\n    Bug: 30950746\n\n    Change-Id: Icfad20f1abb77f91068d244fb0d87fa40824dd1b\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 1143b9bdac12c41474b22ce8cee88c9fb2c07fff\nAuthor: Anay Wadhera \u003cawadhera@berkeley.edu\u003e\nDate:   Fri May 21 13:29:12 2021 -0700\n\n    bpf: move bpf_map_show_fdinfo to match upstream location\n\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 444a9ce4813c9f9347fb7b9a5d227b36fa97cc75\nAuthor: Edward Cree \u003cecree@solarflare.com\u003e\nDate:   Fri Sep 15 14:37:38 2017 +0100\n\n    bpf/verifier: reject BPF_ALU64|BPF_END\n\n    [ Upstream commit e67b8a685c7c984e834e3181ef4619cd7025a136 ]\n\n    Neither ___bpf_prog_run nor the JITs accept it.\n    Also adds a new test case.\n\n    Fixes: 17a5267067f3 (\"bpf: verifier (add verifier core)\")\n    Signed-off-by: Edward Cree \u003cecree@solarflare.com\u003e\n    Acked-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Acked-by: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Greg Kroah-Hartman \u003cgregkh@linuxfoundation.org\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 64115a7394656935b6e032ee22f1637aa5599590\nAuthor: Edward Cree \u003cecree@solarflare.com\u003e\nDate:   Fri Jul 21 14:37:34 2017 +0100\n\n    bpf/verifier: fix min/max handling in BPF_SUB\n\n    [ Upstream commit 9305706c2e808ae59f1eb201867f82f1ddf6d7a6 ]\n\n    We have to subtract the src max from the dst min, and vice-versa, since\n     (e.g.) the smallest result comes from the largest subtrahend.\n\n    Fixes: 484611357c19 (\"bpf: allow access into map value arrays\")\n    Signed-off-by: Edward Cree \u003cecree@solarflare.com\u003e\n    Acked-by: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Greg Kroah-Hartman \u003cgregkh@linuxfoundation.org\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 9fd0c564e2a8dd1db92714c650c8bbc4d44a61c2\nAuthor: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\nDate:   Fri Jul 21 00:00:21 2017 +0200\n\n    bpf: fix mixed signed/unsigned derived min/max value bounds\n\n    [ Upstream commit 4cabc5b186b5427b9ee5a7495172542af105f02b ]\n\n    Edward reported that there\u0027s an issue in min/max value bounds\n    tracking when signed and unsigned compares both provide hints\n    on limits when having unknown variables. E.g. a program such\n    as the following should have been rejected:\n\n       0: (7a) *(u64 *)(r10 -8) \u003d 0\n       1: (bf) r2 \u003d r10\n       2: (07) r2 +\u003d -8\n       3: (18) r1 \u003d 0xffff8a94cda93400\n       5: (85) call bpf_map_lookup_elem#1\n       6: (15) if r0 \u003d\u003d 0x0 goto pc+7\n      R0\u003dmap_value(ks\u003d8,vs\u003d8,id\u003d0),min_value\u003d0,max_value\u003d0 R10\u003dfp\n       7: (7a) *(u64 *)(r10 -16) \u003d -8\n       8: (79) r1 \u003d *(u64 *)(r10 -16)\n       9: (b7) r2 \u003d -1\n      10: (2d) if r1 \u003e r2 goto pc+3\n      R0\u003dmap_value(ks\u003d8,vs\u003d8,id\u003d0),min_value\u003d0,max_value\u003d0 R1\u003dinv,min_value\u003d0\n      R2\u003dimm-1,max_value\u003d18446744073709551615,min_align\u003d1 R10\u003dfp\n      11: (65) if r1 s\u003e 0x1 goto pc+2\n      R0\u003dmap_value(ks\u003d8,vs\u003d8,id\u003d0),min_value\u003d0,max_value\u003d0 R1\u003dinv,min_value\u003d0,max_value\u003d1\n      R2\u003dimm-1,max_value\u003d18446744073709551615,min_align\u003d1 R10\u003dfp\n      12: (0f) r0 +\u003d r1\n      13: (72) *(u8 *)(r0 +0) \u003d 0\n      R0\u003dmap_value_adj(ks\u003d8,vs\u003d8,id\u003d0),min_value\u003d0,max_value\u003d1 R1\u003dinv,min_value\u003d0,max_value\u003d1\n      R2\u003dimm-1,max_value\u003d18446744073709551615,min_align\u003d1 R10\u003dfp\n      14: (b7) r0 \u003d 0\n      15: (95) exit\n\n    What happens is that in the first part ...\n\n       8: (79) r1 \u003d *(u64 *)(r10 -16)\n       9: (b7) r2 \u003d -1\n      10: (2d) if r1 \u003e r2 goto pc+3\n\n    ... r1 carries an unsigned value, and is compared as unsigned\n    against a register carrying an immediate. Verifier deduces in\n    reg_set_min_max() that since the compare is unsigned and operation\n    is greater than (\u003e), that in the fall-through/false case, r1\u0027s\n    minimum bound must be 0 and maximum bound must be r2. Latter is\n    larger than the bound and thus max value is reset back to being\n    \u0027invalid\u0027 aka BPF_REGISTER_MAX_RANGE. Thus, r1 state is now\n    \u0027R1\u003dinv,min_value\u003d0\u0027. The subsequent test ...\n\n      11: (65) if r1 s\u003e 0x1 goto pc+2\n\n    ... is a signed compare of r1 with immediate value 1. Here,\n    verifier deduces in reg_set_min_max() that since the compare\n    is signed this time and operation is greater than (\u003e), that\n    in the fall-through/false case, we can deduce that r1\u0027s maximum\n    bound must be 1, meaning with prior test, we result in r1 having\n    the following state: R1\u003dinv,min_value\u003d0,max_value\u003d1. Given that\n    the actual value this holds is -8, the bounds are wrongly deduced.\n    When this is being added to r0 which holds the map_value(_adj)\n    type, then subsequent store access in above case will go through\n    check_mem_access() which invokes check_map_access_adj(), that\n    will then probe whether the map memory is in bounds based\n    on the min_value and max_value as well as access size since\n    the actual unknown value is min_value \u003c\u003d x \u003c\u003d max_value; commit\n    fce366a9dd0d (\"bpf, verifier: fix alu ops against map_value{,\n    _adj} register types\") provides some more explanation on the\n    semantics.\n\n    It\u0027s worth to note in this context that in the current code,\n    min_value and max_value tracking are used for two things, i)\n    dynamic map value access via check_map_access_adj() and since\n    commit 06c1c049721a (\"bpf: allow helpers access to variable memory\")\n    ii) also enforced at check_helper_mem_access() when passing a\n    memory address (pointer to packet, map value, stack) and length\n    pair to a helper and the length in this case is an unknown value\n    defining an access range through min_value/max_value in that\n    case. The min_value/max_value tracking is /not/ used in the\n    direct packet access case to track ranges. However, the issue\n    also affects case ii), for example, the following crafted program\n    based on the same principle must be rejected as well:\n\n       0: (b7) r2 \u003d 0\n       1: (bf) r3 \u003d r10\n       2: (07) r3 +\u003d -512\n       3: (7a) *(u64 *)(r10 -16) \u003d -8\n       4: (79) r4 \u003d *(u64 *)(r10 -16)\n       5: (b7) r6 \u003d -1\n       6: (2d) if r4 \u003e r6 goto pc+5\n      R1\u003dctx R2\u003dimm0,min_value\u003d0,max_value\u003d0,min_align\u003d2147483648 R3\u003dfp-512\n      R4\u003dinv,min_value\u003d0 R6\u003dimm-1,max_value\u003d18446744073709551615,min_align\u003d1 R10\u003dfp\n       7: (65) if r4 s\u003e 0x1 goto pc+4\n      R1\u003dctx R2\u003dimm0,min_value\u003d0,max_value\u003d0,min_align\u003d2147483648 R3\u003dfp-512\n      R4\u003dinv,min_value\u003d0,max_value\u003d1 R6\u003dimm-1,max_value\u003d18446744073709551615,min_align\u003d1\n      R10\u003dfp\n       8: (07) r4 +\u003d 1\n       9: (b7) r5 \u003d 0\n      10: (6a) *(u16 *)(r10 -512) \u003d 0\n      11: (85) call bpf_skb_load_bytes#26\n      12: (b7) r0 \u003d 0\n      13: (95) exit\n\n    Meaning, while we initialize the max_value stack slot that the\n    verifier thinks we access in the [1,2] range, in reality we\n    pass -7 as length which is interpreted as u32 in the helper.\n    Thus, this issue is relevant also for the case of helper ranges.\n    Resetting both bounds in check_reg_overflow() in case only one\n    of them exceeds limits is also not enough as similar test can be\n    created that uses values which are within range, thus also here\n    learned min value in r1 is incorrect when mixed with later signed\n    test to create a range:\n\n       0: (7a) *(u64 *)(r10 -8) \u003d 0\n       1: (bf) r2 \u003d r10\n       2: (07) r2 +\u003d -8\n       3: (18) r1 \u003d 0xffff880ad081fa00\n       5: (85) call bpf_map_lookup_elem#1\n       6: (15) if r0 \u003d\u003d 0x0 goto pc+7\n      R0\u003dmap_value(ks\u003d8,vs\u003d8,id\u003d0),min_value\u003d0,max_value\u003d0 R10\u003dfp\n       7: (7a) *(u64 *)(r10 -16) \u003d -8\n       8: (79) r1 \u003d *(u64 *)(r10 -16)\n       9: (b7) r2 \u003d 2\n      10: (3d) if r2 \u003e\u003d r1 goto pc+3\n      R0\u003dmap_value(ks\u003d8,vs\u003d8,id\u003d0),min_value\u003d0,max_value\u003d0 R1\u003dinv,min_value\u003d3\n      R2\u003dimm2,min_value\u003d2,max_value\u003d2,min_align\u003d2 R10\u003dfp\n      11: (65) if r1 s\u003e 0x4 goto pc+2\n      R0\u003dmap_value(ks\u003d8,vs\u003d8,id\u003d0),min_value\u003d0,max_value\u003d0\n      R1\u003dinv,min_value\u003d3,max_value\u003d4 R2\u003dimm2,min_value\u003d2,max_value\u003d2,min_align\u003d2 R10\u003dfp\n      12: (0f) r0 +\u003d r1\n      13: (72) *(u8 *)(r0 +0) \u003d 0\n      R0\u003dmap_value_adj(ks\u003d8,vs\u003d8,id\u003d0),min_value\u003d3,max_value\u003d4\n      R1\u003dinv,min_value\u003d3,max_value\u003d4 R2\u003dimm2,min_value\u003d2,max_value\u003d2,min_align\u003d2 R10\u003dfp\n      14: (b7) r0 \u003d 0\n      15: (95) exit\n\n    This leaves us with two options for fixing this: i) to invalidate\n    all prior learned information once we switch signed context, ii)\n    to track min/max signed and unsigned boundaries separately as\n    done in [0]. (Given latter introduces major changes throughout\n    the whole verifier, it\u0027s rather net-next material, thus this\n    patch follows option i), meaning we can derive bounds either\n    from only signed tests or only unsigned tests.) There is still the\n    case of adjust_reg_min_max_vals(), where we adjust bounds on ALU\n    operations, meaning programs like the following where boundaries\n    on the reg get mixed in context later on when bounds are merged\n    on the dst reg must get rejected, too:\n\n       0: (7a) *(u64 *)(r10 -8) \u003d 0\n       1: (bf) r2 \u003d r10\n       2: (07) r2 +\u003d -8\n       3: (18) r1 \u003d 0xffff89b2bf87ce00\n       5: (85) call bpf_map_lookup_elem#1\n       6: (15) if r0 \u003d\u003d 0x0 goto pc+6\n      R0\u003dmap_value(ks\u003d8,vs\u003d8,id\u003d0),min_value\u003d0,max_value\u003d0 R10\u003dfp\n       7: (7a) *(u64 *)(r10 -16) \u003d -8\n       8: (79) r1 \u003d *(u64 *)(r10 -16)\n       9: (b7) r2 \u003d 2\n      10: (3d) if r2 \u003e\u003d r1 goto pc+2\n      R0\u003dmap_value(ks\u003d8,vs\u003d8,id\u003d0),min_value\u003d0,max_value\u003d0 R1\u003dinv,min_value\u003d3\n      R2\u003dimm2,min_value\u003d2,max_value\u003d2,min_align\u003d2 R10\u003dfp\n      11: (b7) r7 \u003d 1\n      12: (65) if r7 s\u003e 0x0 goto pc+2\n      R0\u003dmap_value(ks\u003d8,vs\u003d8,id\u003d0),min_value\u003d0,max_value\u003d0 R1\u003dinv,min_value\u003d3\n      R2\u003dimm2,min_value\u003d2,max_value\u003d2,min_align\u003d2 R7\u003dimm1,max_value\u003d0 R10\u003dfp\n      13: (b7) r0 \u003d 0\n      14: (95) exit\n\n      from 12 to 15: R0\u003dmap_value(ks\u003d8,vs\u003d8,id\u003d0),min_value\u003d0,max_value\u003d0\n      R1\u003dinv,min_value\u003d3 R2\u003dimm2,min_value\u003d2,max_value\u003d2,min_align\u003d2 R7\u003dimm1,min_value\u003d1 R10\u003dfp\n      15: (0f) r7 +\u003d r1\n      16: (65) if r7 s\u003e 0x4 goto pc+2\n      R0\u003dmap_value(ks\u003d8,vs\u003d8,id\u003d0),min_value\u003d0,max_value\u003d0 R1\u003dinv,min_value\u003d3\n      R2\u003dimm2,min_value\u003d2,max_value\u003d2,min_align\u003d2 R7\u003dinv,min_value\u003d4,max_value\u003d4 R10\u003dfp\n      17: (0f) r0 +\u003d r7\n      18: (72) *(u8 *)(r0 +0) \u003d 0\n      R0\u003dmap_value_adj(ks\u003d8,vs\u003d8,id\u003d0),min_value\u003d4,max_value\u003d4 R1\u003dinv,min_value\u003d3\n      R2\u003dimm2,min_value\u003d2,max_value\u003d2,min_align\u003d2 R7\u003dinv,min_value\u003d4,max_value\u003d4 R10\u003dfp\n      19: (b7) r0 \u003d 0\n      20: (95) exit\n\n    Meaning, in adjust_reg_min_max_vals() we must also reset range\n    values on the dst when src/dst registers have mixed signed/\n    unsigned derived min/max value bounds with one unbounded value\n    as otherwise they can be added together deducing false boundaries.\n    Once both boundaries are established from either ALU ops or\n    compare operations w/o mixing signed/unsigned insns, then they\n    can safely be added to other regs also having both boundaries\n    established. Adding regs with one unbounded side to a map value\n    where the bounded side has been learned w/o mixing ops is\n    possible, but the resulting map value won\u0027t recover from that,\n    meaning such op is considered invalid on the time of actual\n    access. Invalid bounds are set on the dst reg in case i) src reg,\n    or ii) in case dst reg already had them. The only way to recover\n    would be to perform i) ALU ops but only \u0027add\u0027 is allowed on map\n    value types or ii) comparisons, but these are disallowed on\n    pointers in case they span a range. This is fine as only BPF_JEQ\n    and BPF_JNE may be performed on PTR_TO_MAP_VALUE_OR_NULL registers\n    which potentially turn them into PTR_TO_MAP_VALUE type depending\n    on the branch, so only here min/max value cannot be invalidated\n    for them.\n\n    In terms of state pruning, value_from_signed is considered\n    as well in states_equal() when dealing with adjusted map values.\n    With regards to breaking existing programs, there is a small\n    risk, but use-cases are rather quite narrow where this could\n    occur and mixing compares probably unlikely.\n\n    Joint work with Josef and Edward.\n\n      [0] https://lists.iovisor.org/pipermail/iovisor-dev/2017-June/000822.html\n\n    Fixes: 484611357c19 (\"bpf: allow access into map value arrays\")\n    Reported-by: Edward Cree \u003cecree@solarflare.com\u003e\n    Signed-off-by: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Signed-off-by: Edward Cree \u003cecree@solarflare.com\u003e\n    Signed-off-by: Josef Bacik \u003cjbacik@fb.com\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Greg Kroah-Hartman \u003cgregkh@linuxfoundation.org\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit c2122c6e7b11f1347cf2cf660ec89050d0f084d5\nAuthor: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\nDate:   Fri Mar 31 02:24:02 2017 +0200\n\n    bpf, verifier: fix alu ops against map_value{, _adj} register types\n\n    [ Upstream commit fce366a9dd0ddc47e7ce05611c266e8574a45116 ]\n\n    While looking into map_value_adj, I noticed that alu operations\n    directly on the map_value() resp. map_value_adj() register (any\n    alu operation on a map_value() register will turn it into a\n    map_value_adj() typed register) are not sufficiently protected\n    against some of the operations. Two non-exhaustive examples are\n    provided that the verifier needs to reject:\n\n     i) BPF_AND on r0 (map_value_adj):\n\n      0: (bf) r2 \u003d r10\n      1: (07) r2 +\u003d -8\n      2: (7a) *(u64 *)(r2 +0) \u003d 0\n      3: (18) r1 \u003d 0xbf842a00\n      5: (85) call bpf_map_lookup_elem#1\n      6: (15) if r0 \u003d\u003d 0x0 goto pc+2\n       R0\u003dmap_value(ks\u003d8,vs\u003d48,id\u003d0),min_value\u003d0,max_value\u003d0 R10\u003dfp\n      7: (57) r0 \u0026\u003d 8\n      8: (7a) *(u64 *)(r0 +0) \u003d 22\n       R0\u003dmap_value_adj(ks\u003d8,vs\u003d48,id\u003d0),min_value\u003d0,max_value\u003d8 R10\u003dfp\n      9: (95) exit\n\n      from 6 to 9: R0\u003dinv,min_value\u003d0,max_value\u003d0 R10\u003dfp\n      9: (95) exit\n      processed 10 insns\n\n    ii) BPF_ADD in 32 bit mode on r0 (map_value_adj):\n\n      0: (bf) r2 \u003d r10\n      1: (07) r2 +\u003d -8\n      2: (7a) *(u64 *)(r2 +0) \u003d 0\n      3: (18) r1 \u003d 0xc24eee00\n      5: (85) call bpf_map_lookup_elem#1\n      6: (15) if r0 \u003d\u003d 0x0 goto pc+2\n       R0\u003dmap_value(ks\u003d8,vs\u003d48,id\u003d0),min_value\u003d0,max_value\u003d0 R10\u003dfp\n      7: (04) (u32) r0 +\u003d (u32) 0\n      8: (7a) *(u64 *)(r0 +0) \u003d 22\n       R0\u003dmap_value_adj(ks\u003d8,vs\u003d48,id\u003d0),min_value\u003d0,max_value\u003d0 R10\u003dfp\n      9: (95) exit\n\n      from 6 to 9: R0\u003dinv,min_value\u003d0,max_value\u003d0 R10\u003dfp\n      9: (95) exit\n      processed 10 insns\n\n    Issue is, while min_value / max_value boundaries for the access\n    are adjusted appropriately, we change the pointer value in a way\n    that cannot be sufficiently tracked anymore from its origin.\n    Operations like BPF_{AND,OR,DIV,MUL,etc} on a destination register\n    that is PTR_TO_MAP_VALUE{,_ADJ} was probably unintended, in fact,\n    all the test cases coming with 484611357c19 (\"bpf: allow access\n    into map value arrays\") perform BPF_ADD only on the destination\n    register that is PTR_TO_MAP_VALUE_ADJ.\n\n    Only for UNKNOWN_VALUE register types such operations make sense,\n    f.e. with unknown memory content fetched initially from a constant\n    offset from the map value memory into a register. That register is\n    then later tested against lower / upper bounds, so that the verifier\n    can then do the tracking of min_value / max_value, and properly\n    check once that UNKNOWN_VALUE register is added to the destination\n    register with type PTR_TO_MAP_VALUE{,_ADJ}. This is also what the\n    original use-case is solving. Note, tracking on what is being\n    added is done through adjust_reg_min_max_vals() and later access\n    to the map value enforced with these boundaries and the given offset\n    from the insn through check_map_access_adj().\n\n    Tests will fail for non-root environment due to prohibited pointer\n    arithmetic, in particular in check_alu_op(), we bail out on the\n    is_pointer_value() check on the dst_reg (which is false in root\n    case as we allow for pointer arithmetic via env-\u003eallow_ptr_leaks).\n\n    Similarly to PTR_TO_PACKET, one way to fix it is to restrict the\n    allowed operations on PTR_TO_MAP_VALUE{,_ADJ} registers to 64 bit\n    mode BPF_ADD. The test_verifier suite runs fine after the patch\n    and it also rejects mentioned test cases.\n\n    Fixes: 484611357c19 (\"bpf: allow access into map value arrays\")\n    Signed-off-by: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Reviewed-by: Josef Bacik \u003cjbacik@fb.com\u003e\n    Acked-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Greg Kroah-Hartman \u003cgregkh@linuxfoundation.org\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit cdc6b46057f4ad4d5090608902c66769671fb5da\nAuthor: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\nDate:   Thu May 18 03:00:06 2017 +0200\n\n    bpf: adjust verifier heuristics\n\n    [ Upstream commit 3c2ce60bdd3d57051bf85615deec04a694473840 ]\n\n    Current limits with regards to processing program paths do not\n    really reflect today\u0027s needs anymore due to programs becoming\n    more complex and verifier smarter, keeping track of more data\n    such as const ALU operations, alignment tracking, spilling of\n    PTR_TO_MAP_VALUE_ADJ registers, and other features allowing for\n    smarter matching of what LLVM generates.\n\n    This also comes with the side-effect that we result in fewer\n    opportunities to prune search states and thus often need to do\n    more work to prove safety than in the past due to different\n    register states and stack layout where we mismatch. Generally,\n    it\u0027s quite hard to determine what caused a sudden increase in\n    complexity, it could be caused by something as trivial as a\n    single branch somewhere at the beginning of the program where\n    LLVM assigned a stack slot that is marked differently throughout\n    other branches and thus causing a mismatch, where verifier\n    then needs to prove safety for the whole rest of the program.\n    Subsequently, programs with even less than half the insn size\n    limit can get rejected. We noticed that while some programs\n    load fine under pre 4.11, they get rejected due to hitting\n    limits on more recent kernels. We saw that in the vast majority\n    of cases (90+%) pruning failed due to register mismatches. In\n    case of stack mismatches, majority of cases failed due to\n    different stack slot types (invalid, spill, misc) rather than\n    differences in spilled registers.\n\n    This patch makes pruning more aggressive by also adding markers\n    that sit at conditional jumps as well. Currently, we only mark\n    jump targets for pruning. For example in direct packet access,\n    these are usually error paths where we bail out. We found that\n    adding these markers, it can reduce number of processed insns\n    by up to 30%. Another option is to ignore reg-\u003eid in probing\n    PTR_TO_MAP_VALUE_OR_NULL registers, which can help pruning\n    slightly as well by up to 7% observed complexity reduction as\n    stand-alone. Meaning, if a previous path with register type\n    PTR_TO_MAP_VALUE_OR_NULL for map X was found to be safe, then\n    in the current state a PTR_TO_MAP_VALUE_OR_NULL register for\n    the same map X must be safe as well. Last but not least the\n    patch also adds a scheduling point and bumps the current limit\n    for instructions to be processed to a more adequate value.\n\n    Signed-off-by: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Acked-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Greg Kroah-Hartman \u003cgregkh@linuxfoundation.org\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit cff4c9704f3eb7fa3797b93ddd32df547bf628a8\nAuthor: John Fastabend \u003cjohn.fastabend@gmail.com\u003e\nDate:   Sun Jul 2 02:13:30 2017 +0200\n\n    bpf, verifier: add additional patterns to evaluate_reg_imm_alu\n\n    [ Upstream commit 43188702b3d98d2792969a3377a30957f05695e6 ]\n\n    Currently the verifier does not track imm across alu operations when\n    the source register is of unknown type. This adds additional pattern\n    matching to catch this and track imm. We\u0027ve seen LLVM generating this\n    pattern while working on cilium.\n\n    Signed-off-by: John Fastabend \u003cjohn.fastabend@gmail.com\u003e\n    Acked-by: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Acked-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Greg Kroah-Hartman \u003cgregkh@linuxfoundation.org\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 87955ac1ac96f5bba169069a33e7a8d0530b5376\nAuthor: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\nDate:   Thu Jun 29 03:04:59 2017 +0200\n\n    bpf: prevent leaking pointer via xadd on unpriviledged\n\n    commit 6bdf6abc56b53103324dfd270a86580306e1a232 upstream.\n\n    Leaking kernel addresses on unpriviledged is generally disallowed,\n    for example, verifier rejects the following:\n\n      0: (b7) r0 \u003d 0\n      1: (18) r2 \u003d 0xffff897e82304400\n      3: (7b) *(u64 *)(r1 +48) \u003d r2\n      R2 leaks addr into ctx\n\n    Doing pointer arithmetic on them is also forbidden, so that they\n    don\u0027t turn into unknown value and then get leaked out. However,\n    there\u0027s xadd as a special case, where we don\u0027t check the src reg\n    for being a pointer register, e.g. the following will pass:\n\n      0: (b7) r0 \u003d 0\n      1: (7b) *(u64 *)(r1 +48) \u003d r0\n      2: (18) r2 \u003d 0xffff897e82304400 ; map\n      4: (db) lock *(u64 *)(r1 +48) +\u003d r2\n      5: (95) exit\n\n    We could store the pointer into skb-\u003ecb, loose the type context,\n    and then read it out from there again to leak it eventually out\n    of a map value. Or more easily in a different variant, too:\n\n       0: (bf) r6 \u003d r1\n       1: (7a) *(u64 *)(r10 -8) \u003d 0\n       2: (bf) r2 \u003d r10\n       3: (07) r2 +\u003d -8\n       4: (18) r1 \u003d 0x0\n       6: (85) call bpf_map_lookup_elem#1\n       7: (15) if r0 \u003d\u003d 0x0 goto pc+3\n       R0\u003dmap_value(ks\u003d8,vs\u003d8,id\u003d0),min_value\u003d0,max_value\u003d0 R6\u003dctx R10\u003dfp\n       8: (b7) r3 \u003d 0\n       9: (7b) *(u64 *)(r0 +0) \u003d r3\n      10: (db) lock *(u64 *)(r0 +0) +\u003d r6\n      11: (b7) r0 \u003d 0\n      12: (95) exit\n\n      from 7 to 11: R0\u003dinv,min_value\u003d0,max_value\u003d0 R6\u003dctx R10\u003dfp\n      11: (b7) r0 \u003d 0\n      12: (95) exit\n\n    Prevent this by checking xadd src reg for pointer types. Also\n    add a couple of test cases related to this.\n\n    Fixes: 1be7f75d1668 (\"bpf: enable non-root eBPF programs\")\n    Fixes: 17a5267067f3 (\"bpf: verifier (add verifier core)\")\n    Signed-off-by: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Acked-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Acked-by: Martin KaFai Lau \u003ckafai@fb.com\u003e\n    Acked-by: Edward Cree \u003cecree@solarflare.com\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Greg Kroah-Hartman \u003cgregkh@linuxfoundation.org\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit a42b58bea9c4a4bac78d576f9a7ef5814e7f4d74\nAuthor: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\nDate:   Wed Jan 18 15:14:17 2017 +0100\n\n    bpf: don\u0027t trigger OOM killer under pressure with map alloc\n\n    [ Upstream commit d407bd25a204bd66b7346dde24bd3d37ef0e0b05 ]\n\n    This patch adds two helpers, bpf_map_area_alloc() and bpf_map_area_free(),\n    that are to be used for map allocations. Using kmalloc() for very large\n    allocations can cause excessive work within the page allocator, so i) fall\n    back earlier to vmalloc() when the attempt is considered costly anyway,\n    and even more importantly ii) don\u0027t trigger OOM killer with any of the\n    allocators.\n\n    Since this is based on a user space request, for example, when creating\n    maps with element pre-allocation, we really want such requests to fail\n    instead of killing other user space processes.\n\n    Also, don\u0027t spam the kernel log with warnings should any of the allocations\n    fail under pressure. Given that, we can make backend selection in\n    bpf_map_area_alloc() generic, and convert all maps over to use this API\n    for spots with potentially large allocation requests.\n\n    Note, replacing the one kmalloc_array() is fine as overflow checks happen\n    earlier in htab_map_alloc(), since it must also protect the multiplication\n    for vmalloc() should kmalloc_array() fail.\n\n    Signed-off-by: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Acked-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Sasha Levin \u003calexander.levin@verizon.com\u003e\n    Signed-off-by: Greg Kroah-Hartman \u003cgregkh@linuxfoundation.org\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit c46236b2dccb66754c132192bbc65719aa0adc87\nAuthor: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\nDate:   Tue Jun 6 18:38:04 2017 +0200\n\n    FROMLIST: bpf: cgroup skb progs cannot access ld_abs/ind\n\n    Commit fb9a307d11d6 (\"bpf: Allow CGROUP_SKB eBPF program to\n    access sk_buff\") enabled programs of BPF_PROG_TYPE_CGROUP_SKB\n    type to use ld_abs/ind instructions. However, at this point,\n    we cannot use them, since offsets relative to SKF_LL_OFF will\n    end up pointing skb_mac_header(skb) out of bounds since in the\n    egress path it is not yet set at that point in time, but only\n    after __dev_queue_xmit() did a general reset on the mac header.\n    bpf_internal_load_pointer_neg_helper() will then end up reading\n    data from a wrong offset.\n\n    BPF_PROG_TYPE_CGROUP_SKB programs can use bpf_skb_load_bytes()\n    already to access packet data, which is also more flexible than\n    the insns carried over from cBPF.\n\n    Fixes: fb9a307d11d6 (\"bpf: Allow CGROUP_SKB eBPF program to access sk_buff\")\n    Signed-off-by: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Acked-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Cc: Chenbo Feng \u003cfengc@google.com\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n\n    (url: http://patchwork.ozlabs.org/patch/771946/)\n\n    Signed-off-by: Chenbo Feng \u003cfengc@google.com\u003e\n    Bug: 30950746\n    Change-Id: Ia32ac79d8c0d18f811ec101897284a8b60cb042a\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit b92174c93e98fda98fcf8bc178cc6bd3bbf50166\nAuthor: Chenbo Feng \u003cfengc@google.com\u003e\nDate:   Fri Jun 2 17:24:31 2017 -0700\n\n    FROMLIST: [net-next,v2,2/2] bpf: Remove the capability check for cgroup skb eBPF program\n\n    Currently loading a cgroup skb eBPF program require a CAP_SYS_ADMIN\n    capability while attaching the program to a cgroup only requires the\n    user have CAP_NET_ADMIN privilege. We can escape the capability\n    check when load the program just like socket filter program to make\n    the capability requirement consistent.\n\n    Change since v1:\n    Change the code style in order to be compliant with checkpatch.pl\n    preference\n\n    (url: http://patchwork.ozlabs.org/patch/769460/)\n    Signed-off-by: Chenbo Feng \u003cfengc@google.com\u003e\n    Bug: 30950746\n    Change-Id: Ibe51235127d6f9349b8f563ad31effc061b278ed\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 46b9aee544752e6342b5c1bb6e1c7d0185932b30\nAuthor: Chenbo Feng \u003cfengc@google.com\u003e\nDate:   Fri Jun 2 17:04:59 2017 -0700\n\n    FROMLIST: [net-next,v2,1/2] bpf: Allow CGROUP_SKB eBPF program to access sk_buff\n\n    This allows cgroup eBPF program to classify packet based on their\n    protocol or other detail information. Currently program need\n    CAP_NET_ADMIN privilege to attach a cgroup eBPF program, and A\n    process with CAP_NET_ADMIN can already see all packets on the system,\n    for example, by creating an iptables rules that causes the packet to\n    be passed to userspace via NFLOG.\n\n    (url: http://patchwork.ozlabs.org/patch/769459/)\n\n    Signed-off-by: Chenbo Feng \u003cfengc@google.com\u003e\n    Bug: 30950746\n    Change-Id: I11bef84ce26cf8b8f1b89483c32a7fcdd61ae926\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 877b99f82bfe29cf3efaf6b4db1909f39c1d1e05\nAuthor: Daniel Mack \u003cdaniel@zonque.org\u003e\nDate:   Mon Nov 28 14:11:04 2016 +0100\n\n    UPSTREAM: bpf: cgroup: fix documentation of __cgroup_bpf_update()\n\n    There\u0027s a \u0027not\u0027 missing in one paragraph. Add it.\n\n    Fixes: 3007098494be (\"cgroup: add support for eBPF programs\")\n    Signed-off-by: Daniel Mack \u003cdaniel@zonque.org\u003e\n    Reported-by: Rami Rosen \u003croszenrami@gmail.com\u003e\n    Acked-by: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Acked-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n\n    Fixes: Change-Id: I3df35d8d3b1261503f9b5bcd90b18c9358f1ac28\n           (\"cgroup: add support for eBPF programs\")\n    (cherry picked from commit 01ae87eab53675cbdabd5c4d727c4a35e397cce0)\n    Signed-off-by: Amit Pundir \u003camit.pundir@linaro.org\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 26d822ed6770290d98d6e1249fd5e7c7bea8d81b\nAuthor: Alexei Starovoitov \u003cast@fb.com\u003e\nDate:   Fri Feb 10 20:28:24 2017 -0800\n\n    BACKPORT: bpf: introduce BPF_F_ALLOW_OVERRIDE flag\n\n    If BPF_F_ALLOW_OVERRIDE flag is used in BPF_PROG_ATTACH command\n    to the given cgroup the descendent cgroup will be able to override\n    effective bpf program that was inherited from this cgroup.\n    By default it\u0027s not passed, therefore override is disallowed.\n\n    Examples:\n    1.\n    prog X attached to /A with default\n    prog Y fails to attach to /A/B and /A/B/C\n    Everything under /A runs prog X\n\n    2.\n    prog X attached to /A with allow_override.\n    prog Y fails to attach to /A/B with default (non-override)\n    prog M attached to /A/B with allow_override.\n    Everything under /A/B runs prog M only.\n\n    3.\n    prog X attached to /A with allow_override.\n    prog Y fails to attach to /A with default.\n    The user has to detach first to switch the mode.\n\n    In the future this behavior may be extended with a chain of\n    non-overridable programs.\n\n    Also fix the bug where detach from cgroup where nothing is attached\n    was not throwing error. Return ENOENT in such case.\n\n    Add several testcases and adjust libbpf.\n\n    Fixes: 3007098494be (\"cgroup: add support for eBPF programs\")\n    Signed-off-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Acked-by: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Acked-by: Tejun Heo \u003ctj@kernel.org\u003e\n    Acked-by: Daniel Mack \u003cdaniel@zonque.org\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n\n    Fixes: Change-Id: I3df35d8d3b1261503f9b5bcd90b18c9358f1ac28\n           (\"cgroup: add support for eBPF programs\")\n    [AmitP: Refactored original patch for android-4.9 where libbpf sources\n            are in samples/bpf/ and test_cgrp2_attach2, test_cgrp2_sock,\n            and test_cgrp2_sock2 sample tests do not exist.]\n    (cherry picked from commit 7f677633379b4abb3281cdbe7e7006f049305c03)\n    Signed-off-by: Amit Pundir \u003camit.pundir@linaro.org\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 86c8466ad31206543466403c1d2f119c913ba756\nAuthor: Daniel Mack \u003cdaniel@zonque.org\u003e\nDate:   Wed Nov 23 16:52:30 2016 +0100\n\n    UPSTREAM: samples: bpf: add userspace example for attaching eBPF programs to cgroups\n\n    Cherry-pick from commit d8c5b17f2bc0de09fbbfa14d90e8168163a579e7\n\n    Add a simple userpace program to demonstrate the new API to attach eBPF\n    programs to cgroups. This is what it does:\n\n     * Create arraymap in kernel with 4 byte keys and 8 byte values\n\n     * Load eBPF program\n\n       The eBPF program accesses the map passed in to store two pieces of\n       information. The number of invocations of the program, which maps\n       to the number of packets received, is stored to key 0. Key 1 is\n       incremented on each iteration by the number of bytes stored in\n       the skb.\n\n     * Detach any eBPF program previously attached to the cgroup\n\n     * Attach the new program to the cgroup using BPF_PROG_ATTACH\n\n     * Once a second, read map[0] and map[1] to see how many bytes and\n       packets were seen on any socket of tasks in the given cgroup.\n\n    The program takes a cgroup path as 1st argument, and either \"ingress\"\n    or \"egress\" as 2nd. Optionally, \"drop\" can be passed as 3rd argument,\n    which will make the generated eBPF program return 0 instead of 1, so\n    the kernel will drop the packet.\n\n    libbpf gained two new wrappers for the new syscall commands.\n\n    Signed-off-by: Daniel Mack \u003cdaniel@zonque.org\u003e\n    Acked-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Bug: 30950746\n    Change-Id: I011436a755abd62050edd22e47995c166a0bd8a2\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 75e278dcb263baf7be1c0df9f30ffbf7d14867d8\nAuthor: Daniel Mack \u003cdaniel@zonque.org\u003e\nDate:   Wed Nov 23 16:52:25 2016 +0100\n\n    UPSTREAM: bpf: add new prog type for cgroup socket filtering\n\n    Cherry-pick from commit 0e33661de493db325435d565a4a722120ae4cbf3\n\n    This program type is similar to BPF_PROG_TYPE_SOCKET_FILTER, except that\n    it does not allow BPF_LD_[ABS|IND] instructions and hooks up the\n    bpf_skb_load_bytes() helper.\n\n    Programs of this type will be attached to cgroups for network filtering\n    and accounting.\n\n    Signed-off-by: Daniel Mack \u003cdaniel@zonque.org\u003e\n    Acked-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Bug: 30950746\n    Change-Id: I7b9e063d5d7a91da80917c6d353a60b877133752\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 47866a0da034c8616f15490be9b8e48436cb31db\nAuthor: Willem de Bruijn \u003cwillemb@google.com\u003e\nDate:   Tue Apr 11 14:08:08 2017 -0400\n\n    BACKPORT: UPSTREAM: bpf: pass sk to helper functions\n\n    Cherrypick from commit 8f917bba0042f1e3b7693743fbe9782709e936e7\n\n    BPF helper functions access socket fields through skb-\u003esk. This is not\n    set in ingress cgroup and socket filters. The association is only made\n    in skb_set_owner_r once the filter has accepted the packet. Sk is\n    available as socket lookup has taken place.\n\n    Temporarily set skb-\u003esk to sk in these cases.\n\n    Signed-off-by: Willem de Bruijn \u003cwillemb@google.com\u003e\n    Acked-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Bug: 30950746\n    Change-Id: Ifcbcbe2ab2882dc79c56f9707be1d6aef08c7fd3\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 7ddf5e3d1ba853a7a0d4b06d8d5dab148b2d7daa\nAuthor: Daniel Mack \u003cdaniel@zonque.org\u003e\nDate:   Wed Nov 23 16:52:27 2016 +0100\n\n    UPSTREAM: bpf: add BPF_PROG_ATTACH and BPF_PROG_DETACH commands\n\n    Cherry-pick from commit f4324551489e8781d838f941b7aee4208e52e8bf\n\n    Extend the bpf(2) syscall by two new commands, BPF_PROG_ATTACH and\n    BPF_PROG_DETACH which allow attaching and detaching eBPF programs\n    to a target.\n\n    On the API level, the target could be anything that has an fd in\n    userspace, hence the name of the field in union bpf_attr is called\n    \u0027target_fd\u0027.\n\n    When called with BPF_ATTACH_TYPE_CGROUP_INET_{E,IN}GRESS, the target is\n    expected to be a valid file descriptor of a cgroup v2 directory which\n    has the bpf controller enabled. These are the only use-cases\n    implemented by this patch at this point, but more can be added.\n\n    If a program of the given type already exists in the given cgroup,\n    the program is swapped automically, so userspace does not have to drop\n    an existing program first before installing a new one, which would\n    otherwise leave a gap in which no program is attached.\n\n    For more information on the propagation logic to subcgroups, please\n    refer to the bpf cgroup controller implementation.\n\n    The API is guarded by CAP_NET_ADMIN.\n\n    Signed-off-by: Daniel Mack \u003cdaniel@zonque.org\u003e\n    Acked-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Bug: 30950746\n    Change-Id: Iab156859332166835d51e1e6f64e5cb8b81870f2\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit f171cff13aab0f7ad8a541e952f88adf9aaa575f\nAuthor: Tejun Heo \u003ctj@kernel.org\u003e\nDate:   Fri Nov 20 15:55:52 2015 -0500\n\n    kernfs: implement kernfs_walk_and_get()\n\n    Implement kernfs_walk_and_get() which is similar to\n    kernfs_find_and_get() but can walk a path instead of just a name.\n\n    v2: Use strlcpy() instead of strlen() + memcpy() as suggested by\n        David.\n\n    Signed-off-by: Tejun Heo \u003ctj@kernel.org\u003e\n    Acked-by: Greg Kroah-Hartman \u003cgregkh@linuxfoundation.org\u003e\n    Cc: David Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit b1deaa8c54ba4bde0c070cc3d10c4ce88d1e1c85\nAuthor: Johannes Weiner \u003channes@cmpxchg.org\u003e\nDate:   Thu Jan 31 10:41:54 2019 +1100\n\n    BACKPORT: fs: kernfs: add poll file operation\n\n    Patch series \"psi: pressure stall monitors\", v3.\n\n    Android is adopting psi to detect and remedy memory pressure that results\n    in stuttering and decreased responsiveness on mobile devices.\n\n    Psi gives us the stall information, but because we\u0027re dealing with\n    latencies in the millisecond range, periodically reading the pressure\n    files to detect stalls in a timely fashion is not feasible.  Psi also\n    doesn\u0027t aggregate its averages at a high enough frequency right now.\n\n    This patch series extends the psi interface such that users can configure\n    sensitive latency thresholds and use poll() and friends to be notified\n    when these are breached.\n\n    As high-frequency aggregation is costly, it implements an aggregation\n    method that is optimized for fast, short-interval averaging, and makes the\n    aggregation frequency adaptive, such that high-frequency updates only\n    happen while monitored stall events are actively occurring.\n\n    With these patches applied, Android can monitor for, and ward off,\n    mounting memory shortages before they cause problems for the user.  For\n    example, using memory stall monitors in userspace low memory killer daemon\n    (lmkd) we can detect mounting pressure and kill less important processes\n    before device becomes visibly sluggish.  In our memory stress testing psi\n    memory monitors produce roughly 10x less false positives compared to\n    vmpressure signals.  Having ability to specify multiple triggers for the\n    same psi metric allows other parts of Android framework to monitor memory\n    state of the device and act accordingly.\n\n    The new interface is straightforward.  The user opens one of the pressure\n    files for writing and writes a trigger description into the file\n    descriptor that defines the stall state - some or full, and the maximum\n    stall time over a given window of time.  E.g.:\n\n            /* Signal when stall time exceeds 100ms of a 1s window */\n            char trigger[] \u003d \"full 100000 1000000\";\n            fd \u003d open(\"/proc/pressure/memory\");\n            write(fd, trigger, sizeof(trigger));\n            while (poll() \u003e\u003d 0) {\n                    ...\n            }\n            close(fd);\n\n    When the monitored stall state is entered, psi adapts its aggregation\n    frequency according to what the configured time window requires in order\n    to emit event signals in a timely fashion.  Once the stalling subsides,\n    aggregation reverts back to normal.\n\n    The trigger is associated with the open file descriptor.  To stop\n    monitoring, the user only needs to close the file descriptor and the\n    trigger is discarded.\n\n    Patches 1-4 prepare the psi code for polling support.  Patch 5 implements\n    the adaptive polling logic, the pressure growth detection optimized for\n    short intervals, and hooks up write() and poll() on the pressure files.\n\n    The patches were developed in collaboration with Johannes Weiner.\n\n    This patch (of 5):\n\n    Kernfs has a standardized poll/notification mechanism for waking all\n    pollers on all fds when a filesystem node changes.  To allow polling for\n    custom events, add a .poll callback that can override the default.\n\n    This is in preparation for pollable cgroup pressure files which have\n    per-fd trigger configurations.\n\n    Link: http://lkml.kernel.org/r/20190124211518.244221-2-surenb@google.com\n    Signed-off-by: Johannes Weiner \u003channes@cmpxchg.org\u003e\n    Signed-off-by: Suren Baghdasaryan \u003csurenb@google.com\u003e\n    Cc: Dennis Zhou \u003cdennis@kernel.org\u003e\n    Cc: Ingo Molnar \u003cmingo@redhat.com\u003e\n    Cc: Jens Axboe \u003caxboe@kernel.dk\u003e\n    Cc: Li Zefan \u003clizefan@huawei.com\u003e\n    Cc: Peter Zijlstra \u003cpeterz@infradead.org\u003e\n    Cc: Tejun Heo \u003ctj@kernel.org\u003e\n    Signed-off-by: Andrew Morton \u003cakpm@linux-foundation.org\u003e\n    Signed-off-by: Stephen Rothwell \u003csfr@canb.auug.org.au\u003e\n\n    (cherry picked from commit: 147e1a97c4a0bdd43f55a582a9416bb9092563a9)\n\n    Conflicts:\n            fs/kernfs/file.c\n            include/linux/kernfs.h\n\n    1. replaced __poll_t with unsigned int.\n    2. replaced kernfs_dentry_node() with dentry-\u003ed_fsdata\n    3. replaced EPOLLERR/EPOLLPRI with POLLERR/POLLPRI (values are the same)\n\n    Bug: 127712811\n    Test: lmkd in PSI mode\n    Change-Id: Ic2bed334d05aec62f4e695f263893c3057921c55\n    Signed-off-by: Suren Baghdasaryan \u003csurenb@google.com\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit a9e35ac00d90adae5c6125c0c2932b3d5527223e\nAuthor: Tejun Heo \u003ctj@kernel.org\u003e\nDate:   Tue Dec 27 14:49:03 2016 -0500\n\n    UPSTREAM: kernfs: add kernfs_ops-\u003eopen/release() callbacks\n\n    Add -\u003eopen/release() methods to kernfs_ops.  -\u003eopen() is called when\n    the file is opened and -\u003erelease() when the file is either released or\n    severed.  These callbacks can be used, for example, to manage\n    persistent caching objects over multiple seq_file iterations.\n\n    Signed-off-by: Tejun Heo \u003ctj@kernel.org\u003e\n    Acked-by: Greg Kroah-Hartman \u003cgregkh@linuxfoundation.org\u003e\n    Acked-by: Acked-by: Zefan Li \u003clizefan@huawei.com\u003e\n\n    (cherry picked from commit 0e67db2f9fe91937e798e3d7d22c50a8438187e1)\n\n    Bug: 111308141\n    Test: modified lmkd to use PSI and tested using lmkd_unit_test\n\n    Change-Id: Id06e9d5c6da1280bcdd4dc86309dcfaf52b8f9a4\n    Signed-off-by: Suren Baghdasaryan \u003csurenb@google.com\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit a8e45dc4b2492150c456c6e5d3675bfed82fb7c8\nAuthor: Aditya Kali \u003cadityakali@google.com\u003e\nDate:   Fri Jan 29 02:54:04 2016 -0600\n\n    kernfs: Add API to generate relative kernfs path\n\n    The new function kernfs_path_from_node() generates and returns kernfs\n    path of a given kernfs_node relative to a given parent kernfs_node.\n\n    Signed-off-by: Aditya Kali \u003cadityakali@google.com\u003e\n    Signed-off-by: Serge E. Hallyn \u003cserge.hallyn@canonical.com\u003e\n    Acked-by: Greg Kroah-Hartman \u003cgregkh@linuxfoundation.org\u003e\n    Signed-off-by: Tejun Heo \u003ctj@kernel.org\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 9406f1fa9a9b9df8367ac5ede9958827c00c2311\nAuthor: Aditya Kali \u003cadityakali@google.com\u003e\nDate:   Fri Jan 29 02:54:05 2016 -0600\n\n    sched: new clone flag CLONE_NEWCGROUP for cgroup namespace\n\n    CLONE_NEWCGROUP will be used to create new cgroup namespace.\n\n    Signed-off-by: Aditya Kali \u003cadityakali@google.com\u003e\n    Signed-off-by: Serge Hallyn \u003cserge.hallyn@canonical.com\u003e\n    Signed-off-by: Tejun Heo \u003ctj@kernel.org\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 3bf60db3e565e84cb71817a070479677ff40f7b4\nAuthor: Daniel Mack \u003cdaniel@zonque.org\u003e\nDate:   Wed Nov 23 16:52:26 2016 +0100\n\n    UPSTREAM: cgroup: add support for eBPF programs\n\n    Cherry-pick from commit 3007098494bec614fb55dee7bc0410bb7db5ad18\n\n    This patch adds two sets of eBPF program pointers to struct cgroup.\n    One for such that are directly pinned to a cgroup, and one for such\n    that are effective for it.\n\n    To illustrate the logic behind that, assume the following example\n    cgroup hierarchy.\n\n      A - B - C\n            \\ D - E\n\n    If only B has a program attached, it will be effective for B, C, D\n    and E. If D then attaches a program itself, that will be effective for\n    both D and E, and the program in B will only affect B and C. Only one\n    program of a given type is effective for a cgroup.\n\n    Attaching and detaching programs will be done through the bpf(2)\n    syscall. For now, ingress and egress inet socket filtering are the\n    only supported use-cases.\n\n    Signed-off-by: Daniel Mack \u003cdaniel@zonque.org\u003e\n    Acked-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Bug: 30950746\n    Change-Id: I3df35d8d3b1261503f9b5bcd90b18c9358f1ac28\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 30186d5cc793fdc8e67f210a0eedeec686409623\nAuthor: Tejun Heo \u003ctj@kernel.org\u003e\nDate:   Thu Jan 26 16:47:28 2017 -0500\n\n    cgroup: don\u0027t online subsystems before cgroup_name/path() are operational\n\n    commit 07cd12945551b63ecb1a349d50a6d69d1d6feb4a upstream.\n\n    While refactoring cgroup creation, a5bca2152036 (\"cgroup: factor out\n    cgroup_create() out of cgroup_mkdir()\") incorrectly onlined subsystems\n    before the new cgroup is associated with it kernfs_node.  This is fine\n    for cgroup proper but cgroup_name/path() depend on the associated\n    kernfs_node and if a subsystem makes the new cgroup_subsys_state\n    visible, which they\u0027re allowed to after onlining, it can lead to NULL\n    dereference.\n\n    The current code performs cgroup creation and subsystem onlining in\n    cgroup_create() and cgroup_mkdir() makes the cgroup and subsystems\n    visible afterwards.  There\u0027s no reason to online the subsystems early\n    and we can simply drop cgroup_apply_control_enable() call from\n    cgroup_create() so that the subsystems are onlined and made visible at\n    the same time.\n\n    Signed-off-by: Tejun Heo \u003ctj@kernel.org\u003e\n    Reported-by: Konstantin Khlebnikov \u003ckhlebnikov@yandex-team.ru\u003e\n    Fixes: a5bca2152036 (\"cgroup: factor out cgroup_create() out of cgroup_mkdir()\")\n    Signed-off-by: Greg Kroah-Hartman \u003cgregkh@linuxfoundation.org\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 2a404ba87afe67b5607825edae0c97c5c44a7e4b\nAuthor: Tejun Heo \u003ctj@kernel.org\u003e\nDate:   Thu Sep 29 15:49:40 2016 +0200\n\n    cgroup: fix error handling regressions in proc_cgroup_show() and cgroup_release_agent()\n\n    4c737b41de7f (\"cgroup: make cgroup_path() and friends behave in the\n    style of strlcpy()\") broke error handling in proc_cgroup_show() and\n    cgroup_release_agent() by not handling negative return values from\n    cgroup_path_ns_locked().  Fix it.\n\n    Reported-by: Dan Carpenter \u003cdan.carpenter@oracle.com\u003e\n    Signed-off-by: Tejun Heo \u003ctj@kernel.org\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n    Change-Id: If1dbcdb90f9eefbfc2aa245a8fc4da5b15e23296\n\ncommit c4d030bfe1d3a783f413aa4a1170a652fd6fe4ec\nAuthor: Tejun Heo \u003ctj@kernel.org\u003e\nDate:   Fri Sep 23 16:55:49 2016 -0400\n\n    cgroup: fix invalid controller enable rejections with cgroup namespace\n\n    On the v2 hierarchy, \"cgroup.subtree_control\" rejects controller\n    enables if the cgroup has processes in it.  The enforcement of this\n    logic assumes that the cgroup wouldn\u0027t have any css_sets associated\n    with it if there are no tasks in the cgroup, which is no longer true\n    since a79a908fd2b0 (\"cgroup: introduce cgroup namespaces\").\n\n    When a cgroup namespace is created, it pins the css_set of the\n    creating task to use it as the root css_set of the namespace.  This\n    extra reference stays as long as the namespace is around and makes\n    \"cgroup.subtree_control\" think that the namespace root cgroup is not\n    empty even when it is and thus reject controller enables.\n\n    Fix it by making cgroup_subtree_control() walk and test emptiness of\n    each css_set instead of testing whether the list_head is empty.\n\n    While at it, update the comment of cgroup_task_count() to indicate\n    that the returned value may be higher than the number of tasks, which\n    has always been true due to temporary references and doesn\u0027t break\n    anything.\n\n    Signed-off-by: Tejun Heo \u003ctj@kernel.org\u003e\n    Reported-by: Evgeny Vereshchagin \u003cevvers@ya.ru\u003e\n    Cc: Serge E. Hallyn \u003cserge.hallyn@ubuntu.com\u003e\n    Cc: Aditya Kali \u003cadityakali@google.com\u003e\n    Cc: Eric W. Biederman \u003cebiederm@xmission.com\u003e\n    Cc: stable@vger.kernel.org # v4.6+\n    Fixes: a79a908fd2b0 (\"cgroup: introduce cgroup namespaces\")\n    Link: https://github.com/systemd/systemd/pull/3589#issuecomment-249089541\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit de050f46d87a48549246a2950ea90eee35c35557\nAuthor: Andrey Vagin \u003cavagin@openvz.org\u003e\nDate:   Tue Sep 6 00:47:13 2016 -0700\n\n    kernel: add a helper to get an owning user namespace for a namespace\n\n    Return -EPERM if an owning user namespace is outside of a process\n    current user namespace.\n\n    v2: In a first version ns_get_owner returned ENOENT for init_user_ns.\n        This special cases was removed from this version. There is nothing\n        outside of init_user_ns, so we can return EPERM.\n    v3: rename ns-\u003eget_owner() to ns-\u003eowner(). get_* usually means that it\n    grabs a reference.\n\n    Acked-by: Serge Hallyn \u003cserge@hallyn.com\u003e\n    Signed-off-by: Andrei Vagin \u003cavagin@openvz.org\u003e\n    Signed-off-by: Eric W. Biederman \u003cebiederm@xmission.com\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 5c5eb1b26c8e98052645f8207a3e1e96b46d0560\nAuthor: Seth Forshee \u003cseth.forshee@canonical.com\u003e\nDate:   Wed Sep 23 15:16:04 2015 -0500\n\n    fs: Limit file caps to the user namespace of the super block\n\n    Capability sets attached to files must be ignored except in the\n    user namespaces where the mounter is privileged, i.e. s_user_ns\n    and its descendants. Otherwise a vector exists for gaining\n    privileges in namespaces where a user is not already privileged.\n\n    Add a new helper function, current_in_user_ns(), to test whether a user\n    namespace is the same as or a descendant of another namespace.\n    Use this helper to determine whether a file\u0027s capability set\n    should be applied to the caps constructed during exec.\n\n    --EWB Replaced in_userns with the simpler current_in_userns.\n\n    Acked-by: Serge Hallyn \u003cserge.hallyn@canonical.com\u003e\n    Signed-off-by: Seth Forshee \u003cseth.forshee@canonical.com\u003e\n    Signed-off-by: Eric W. Biederman \u003cebiederm@xmission.com\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 3c33b6a4207145695027cdaa15426ae8d440d48c\nAuthor: Johannes Weiner \u003cjweiner@fb.com\u003e\nDate:   Mon Sep 19 14:44:38 2016 -0700\n\n    cgroup: duplicate cgroup reference when cloning sockets\n\n    When a socket is cloned, the associated sock_cgroup_data is duplicated\n    but not its reference on the cgroup.  As a result, the cgroup reference\n    count will underflow when both sockets are destroyed later on.\n\n    Fixes: bd1060a1d671 (\"sock, cgroup: add sock-\u003esk_cgroup\")\n    Link: http://lkml.kernel.org/r/20160914194846.11153-2-hannes@cmpxchg.org\n    Signed-off-by: Johannes Weiner \u003channes@cmpxchg.org\u003e\n    Acked-by: Tejun Heo \u003ctj@kernel.org\u003e\n    Cc: Michal Hocko \u003cmhocko@suse.cz\u003e\n    Cc: Vladimir Davydov \u003cvdavydov@virtuozzo.com\u003e\n    Cc: \u003cstable@vger.kernel.org\u003e\t[4.5+]\n    Signed-off-by: Andrew Morton \u003cakpm@linux-foundation.org\u003e\n    Signed-off-by: Linus Torvalds \u003ctorvalds@linux-foundation.org\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 38686f066989f9b4112a122201b2ed3728d283de\nAuthor: Tejun Heo \u003ctj@kernel.org\u003e\nDate:   Wed Aug 10 11:23:44 2016 -0400\n\n    cgroup: make cgroup_path() and friends behave in the style of strlcpy()\n\n    cgroup_path() and friends used to format the path from the end and\n    thus the resulting path usually didn\u0027t start at the start of the\n    passed in buffer.  Also, when the buffer was too small, the partial\n    result was truncated from the head rather than tail and there was no\n    way to tell how long the full path would be.  These make the functions\n    less robust and more awkward to use.\n\n    With recent updates to kernfs_path(), cgroup_path() and friends can be\n    made to behave in strlcpy() style.\n\n    * cgroup_path(), cgroup_path_ns[_locked]() and task_cgroup_path() now\n      always return the length of the full path.  If buffer is too small,\n      it contains nul terminated truncated output.\n\n    * All users updated accordingly.\n\n    v2: cgroup_path() usage in kernel/sched/debug.c converted.\n\n    Signed-off-by: Tejun Heo \u003ctj@kernel.org\u003e\n    Acked-by: Greg Kroah-Hartman \u003cgregkh@linuxfoundation.org\u003e\n    Cc: Serge Hallyn \u003cserge.hallyn@ubuntu.com\u003e\n    Cc: Peter Zijlstra \u003cpeterz@infradead.org\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n    Change-Id: I8f16f5cb47c1eae59ad7539ea3fa3af825e0a125\n\ncommit 472093013bb40049d98f8d8766cda69d0e26d91c\nAuthor: Eric W. Biederman \u003cebiederm@xmission.com\u003e\nDate:   Fri Jul 15 06:36:44 2016 -0500\n\n    cgroupns: Only allow creation of hierarchies in the initial cgroup namespace\n\n    Unprivileged users can\u0027t use hierarchies if they create them as they do not\n    have privilieges to the root directory.\n\n    Which means the only thing a hiearchy created by an unprivileged user\n    is good for is expanding the number of cgroup links in every css_set,\n    which is a DOS attack.\n\n    We could allow hierarchies to be created in namespaces in the initial\n    user namespace.  Unfortunately there is only a single namespace for\n    the names of heirarchies, so that is likely to create more confusion\n    than not.\n\n    So do the simple thing and restrict hiearchy creation to the initial\n    cgroup namespace.\n\n    Cc: stable@vger.kernel.org\n    Fixes: a79a908fd2b0 (\"cgroup: introduce cgroup namespaces\")\n    Signed-off-by: \"Eric W. Biederman\" \u003cebiederm@xmission.com\u003e\n    Signed-off-by: Tejun Heo \u003ctj@kernel.org\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit df92cfc300f92018760720903b9d80cfdbcba1db\nAuthor: Eric W. Biederman \u003cebiederm@xmission.com\u003e\nDate:   Fri Jul 15 06:35:24 2016 -0500\n\n    cgroupns: Fix the locking in copy_cgroup_ns\n\n    If \"clone(CLONE_NEWCGROUP...)\" is called it results in a nice lockdep\n    valid splat.\n\n    In __cgroup_proc_write the lock ordering is:\n         cgroup_mutex -- through cgroup_kn_lock_live\n         cgroup_threadgroup_rwsem\n\n    In copy_process the guts of clone the lock ordering is:\n         cgroup_threadgroup_rwsem -- through threadgroup_change_begin\n         cgroup_mutex -- through copy_namespaces -- copy_cgroup_ns\n\n    lockdep reports some a different call chains for the first ordering of\n    cgroup_mutex and cgroup_threadgroup_rwsem but it is harder to trace.\n    This is most definitely deadlock potential under the right\n    circumstances.\n\n    Fix this by by skipping the cgroup_mutex and making the locking in\n    copy_cgroup_ns mirror the locking in cgroup_post_fork which also runs\n    during fork under the cgroup_threadgroup_rwsem.\n\n    Cc: stable@vger.kernel.org\n    Fixes: a79a908fd2b0 (\"cgroup: introduce cgroup namespaces\")\n    Signed-off-by: \"Eric W. Biederman\" \u003cebiederm@xmission.com\u003e\n    Signed-off-by: Tejun Heo \u003ctj@kernel.org\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit d7f5497338b07f8c3a87a6a516dff87bc3473860\nAuthor: Martin KaFai Lau \u003ckafai@fb.com\u003e\nDate:   Thu Jun 30 10:28:42 2016 -0700\n\n    cgroup: Add cgroup_get_from_fd\n\n    Add a helper function to get a cgroup2 from a fd.  It will be\n    stored in a bpf array (BPF_MAP_TYPE_CGROUP_ARRAY) which will\n    be introduced in the later patch.\n\n    Signed-off-by: Martin KaFai Lau \u003ckafai@fb.com\u003e\n    Cc: Alexei Starovoitov \u003cast@fb.com\u003e\n    Cc: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Cc: Tejun Heo \u003ctj@kernel.org\u003e\n    Acked-by: Tejun Heo \u003ctj@kernel.org\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 0c524dc1e87f8d69651d8d8832680df76cc2225f\nAuthor: Tejun Heo \u003ctj@kernel.org\u003e\nDate:   Tue Jun 21 13:06:24 2016 -0400\n\n    cgroup: allow NULL return from ss-\u003ecss_alloc()\n\n    cgroup core expected css_alloc to return an ERR_PTR value on failure\n    and caused NULL deref if it returned NULL.  It\u0027s an easy mistake to\n    make from an alloc function and there\u0027s no ambiguity in what\u0027s being\n    indicated.  Update css_create() so that it interprets NULL return from\n    css_alloc as -ENOMEM.\n\n    Signed-off-by: Tejun Heo \u003ctj@kernel.org\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 6b65d1f5a27b195a2a99bcb75a4cf8e7059c9afa\nAuthor: Johannes Weiner \u003channes@cmpxchg.org\u003e\nDate:   Fri Jun 17 12:24:27 2016 -0400\n\n    cgroup: remove unnecessary 0 check from css_from_id()\n\n    css_idr allocation starts at 1, so index 0 will never point to an\n    item. css_from_id() currently filters that before asking idr_find(),\n    but idr_find() would also just return NULL, so this is not needed.\n\n    Signed-off-by: Johannes Weiner \u003channes@cmpxchg.org\u003e\n    Signed-off-by: Tejun Heo \u003ctj@kernel.org\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit c273892d57e8195c22d16d978896c220988fd771\nAuthor: Johannes Weiner \u003channes@cmpxchg.org\u003e\nDate:   Fri Jun 17 12:23:59 2016 -0400\n\n    cgroup: fix idr leak for the first cgroup root\n\n    The valid cgroup hierarchy ID range includes 0, so we can\u0027t filter for\n    positive numbers when freeing it, or it\u0027ll leak the first ID. No big\n    deal, just disruptive when reading the code.\n\n    The ID is freed during error handling and when the reference count\n    hits zero, so the double-free test is not necessary; remove it.\n\n    Signed-off-by: Johannes Weiner \u003channes@cmpxchg.org\u003e\n    Signed-off-by: Tejun Heo \u003ctj@kernel.org\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 728fa50652b1a914123400118b02135833083ce5\nAuthor: Wenwei Tao \u003cww.tao0320@gmail.com\u003e\nDate:   Fri May 13 22:59:20 2016 +0800\n\n    cgroup: remove redundant cleanup in css_create\n\n    When create css failed, before call css_free_rcu_fn, we remove the css\n    id and exit the percpu_ref, but we will do these again in\n    css_free_work_fn, so they are redundant.  Especially the css id, that\n    would cause problem if we remove it twice, since it may be assigned to\n    another css after the first remove.\n\n    tj: This was broken by two commits updating the free path without\n        synchronizing the creation failure path.  This can be easily\n        triggered by trying to create more than 64k memory cgroups.\n\n    Signed-off-by: Wenwei Tao \u003cww.tao0320@gmail.com\u003e\n    Signed-off-by: Tejun Heo \u003ctj@kernel.org\u003e\n    Cc: Vladimir Davydov \u003cvdavydov@parallels.com\u003e\n    Fixes: 9a1049da9bd2 (\"percpu-refcount: require percpu_ref to be exited explicitly\")\n    Fixes: 01e586598b22 (\"cgroup: release css-\u003eid after css_free\")\n    Cc: stable@vger.kernel.org # v3.17+\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 444518a90f350094f0fc239588af657581cc2ddd\nAuthor: Felipe Balbi \u003cfelipe.balbi@linux.intel.com\u003e\nDate:   Thu May 12 12:34:38 2016 +0300\n\n    cgroup: fix compile warning\n\n    commit 4f41fc59620f (\"cgroup, kernfs: make mountinfo\n     show properly scoped path for cgroup namespaces\")\n     added the following compile warning:\n\n    kernel/cgroup.c: In function ‘cgroup_show_path’:\n    kernel/cgroup.c:1634:15: warning: unused variable ‘ret’ [-Wunused-variable]\n      int len \u003d 0, ret \u003d 0;\n                   ^\n    fix it.\n\n    Fixes: 4f41fc59620f (\"cgroup, kernfs: make mountinfo show properly scoped path for cgroup namespaces\")\n    Signed-off-by: Felipe Balbi \u003cfelipe.balbi@linux.intel.com\u003e\n    Signed-off-by: Tejun Heo \u003ctj@kernel.org\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit de4ffec830e1ffb5a1e4d7472c223fa2dc01b3d1\nAuthor: Serge E. Hallyn \u003cserge.hallyn@ubuntu.com\u003e\nDate:   Mon May 9 09:59:55 2016 -0500\n\n    cgroup, kernfs: make mountinfo show properly scoped path for cgroup namespaces\n\n    Patch summary:\n\n    When showing a cgroupfs entry in mountinfo, show the path of the mount\n    root dentry relative to the reader\u0027s cgroup namespace root.\n\n    Short explanation (courtesy of mkerrisk):\n\n    If we create a new cgroup namespace, then we want both /proc/self/cgroup\n    and /proc/self/mountinfo to show cgroup paths that are correctly\n    virtualized with respect to the cgroup mount point.  Previous to this\n    patch, /proc/self/cgroup shows the right info, but /proc/self/mountinfo\n    does not.\n\n    Long version:\n\n    When a uid 0 task which is in freezer cgroup /a/b, unshares a new cgroup\n    namespace, and then mounts a new instance of the freezer cgroup, the new\n    mount will be rooted at /a/b.  The root dentry field of the mountinfo\n    entry will show \u0027/a/b\u0027.\n\n     cat \u003e /tmp/do1 \u003c\u003c EOF\n     mount -t cgroup -o freezer freezer /mnt\n     grep freezer /proc/self/mountinfo\n     EOF\n\n     unshare -Gm  bash /tmp/do1\n     \u003e 330 160 0:34 / /sys/fs/cgroup/freezer rw,nosuid,nodev,noexec,relatime - cgroup cgroup rw,freezer\n     \u003e 355 133 0:34 /a/b /mnt rw,relatime - cgroup freezer rw,freezer\n\n    The task\u0027s freezer cgroup entry in /proc/self/cgroup will simply show\n    \u0027/\u0027:\n\n     grep freezer /proc/self/cgroup\n     9:freezer:/\n\n    If instead the same task simply bind mounts the /a/b cgroup directory,\n    the resulting mountinfo entry will again show /a/b for the dentry root.\n    However in this case the task will find its own cgroup at /mnt/a/b,\n    not at /mnt:\n\n     mount --bind /sys/fs/cgroup/freezer/a/b /mnt\n     130 25 0:34 /a/b /mnt rw,nosuid,nodev,noexec,relatime shared:21 - cgroup cgroup rw,freezer\n\n    In other words, there is no way for the task to know, based on what is\n    in mountinfo, which cgroup directory is its own.\n\n    Example (by mkerrisk):\n\n    First, a little script to save some typing and verbiage:\n\n    echo -e \"\\t/proc/self/cgroup:\\t$(cat /proc/self/cgroup | grep freezer)\"\n    cat /proc/self/mountinfo | grep freezer |\n            awk \u0027{print \"\\tmountinfo:\\t\\t\" $4 \"\\t\" $5}\u0027\n\n    Create cgroup, place this shell into the cgroup, and look at the state\n    of the /proc files:\n\n    2653\n    2653                         # Our shell\n    14254                        # cat(1)\n            /proc/self/cgroup:      10:freezer:/a/b\n            mountinfo:              /       /sys/fs/cgroup/freezer\n\n    Create a shell in new cgroup and mount namespaces. The act of creating\n    a new cgroup namespace causes the process\u0027s current cgroups directories\n    to become its cgroup root directories. (Here, I\u0027m using my own version\n    of the \"unshare\" utility, which takes the same options as the util-linux\n    version):\n\n    Look at the state of the /proc files:\n\n            /proc/self/cgroup:      10:freezer:/\n            mountinfo:              /       /sys/fs/cgroup/freezer\n\n    The third entry in /proc/self/cgroup (the pathname of the cgroup inside\n    the hierarchy) is correctly virtualized w.r.t. the cgroup namespace, which\n    is rooted at /a/b in the outer namespace.\n\n    However, the info in /proc/self/mountinfo is not for this cgroup\n    namespace, since we are seeing a duplicate of the mount from the\n    old mount namespace, and the info there does not correspond to the\n    new cgroup namespace. However, trying to create a new mount still\n    doesn\u0027t show us the right information in mountinfo:\n\n                                          # propagating to other mountns\n            /proc/self/cgroup:      7:freezer:/\n            mountinfo:              /a/b    /mnt/freezer\n\n    The act of creating a new cgroup namespace caused the process\u0027s\n    current freezer directory, \"/a/b\", to become its cgroup freezer root\n    directory. In other words, the pathname directory of the directory\n    within the newly mounted cgroup filesystem should be \"/\",\n    but mountinfo wrongly shows us \"/a/b\". The consequence of this is\n    that the process in the cgroup namespace cannot correctly construct\n    the pathname of its cgroup root directory from the information in\n    /proc/PID/mountinfo.\n\n    With this patch, the dentry root field in mountinfo is shown relative\n    to the reader\u0027s cgroup namespace.  So the same steps as above:\n\n            /proc/self/cgroup:      10:freezer:/a/b\n            mountinfo:              /       /sys/fs/cgroup/freezer\n            /proc/self/cgroup:      10:freezer:/\n            mountinfo:              /../..  /sys/fs/cgroup/freezer\n            /proc/self/cgroup:      10:freezer:/\n            mountinfo:              /       /mnt/freezer\n\n    cgroup.clone_children  freezer.parent_freezing  freezer.state      tasks\n    cgroup.procs           freezer.self_freezing    notify_on_release\n    3164\n    2653                   # First shell that placed in this cgroup\n    3164                   # Shell started by \u0027unshare\u0027\n    14197                  # cat(1)\n\n    Signed-off-by: Serge Hallyn \u003cserge.hallyn@ubuntu.com\u003e\n    Tested-by: Michael Kerrisk \u003cmtk.manpages@gmail.com\u003e\n    Acked-by: Michael Kerrisk \u003cmtk.manpages@gmail.com\u003e\n    Signed-off-by: Tejun Heo \u003ctj@kernel.org\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit dbd9d594afbbf236c4c8a299c90492b769774664\nAuthor: Tejun Heo \u003ctj@kernel.org\u003e\nDate:   Tue Mar 8 11:51:26 2016 -0500\n\n    cgroup: implement cgroup_subsys-\u003eimplicit_on_dfl\n\n    Some controllers, perf_event for now and possibly freezer in the\n    future, don\u0027t really make sense to control explicitly through\n    \"cgroup.subtree_control\".  For example, the primary role of perf_event\n    is identifying the cgroups of tasks; however, because the controller\n    also keeps a small amount of state per cgroup, it can\u0027t be replaced\n    with simple cgroup membership tests.\n\n    This patch implements cgroup_subsys-\u003eimplicit_on_dfl flag.  When set,\n    the controller is implicitly enabled on all cgroups on the v2\n    hierarchy so that utility type controllers such as perf_event can be\n    enabled and function transparently.\n\n    An implicit controller doesn\u0027t show up in \"cgroup.controllers\" or\n    \"cgroup.subtree_control\", is exempt from no internal process rule and\n    can be stolen from the default hierarchy even if there are non-root\n    csses.\n\n    v2: Reimplemented on top of the recent updates to css handling and\n        subsystem rebinding.  Rebinding implicit subsystems is now a\n        simple matter of exempting it from the busy subsystem check.\n\n    Signed-off-by: Tejun Heo \u003ctj@kernel.org\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit cbce3ecdb3855ce9062be58392f31f5c16443757\nAuthor: Tejun Heo \u003ctj@kernel.org\u003e\nDate:   Tue Mar 8 11:51:26 2016 -0500\n\n    cgroup: use css_set-\u003emg_dst_cgrp for the migration target cgroup\n\n    Migration can be multi-target on the default hierarchy when a\n    controller is enabled - processes belonging to each child cgroup have\n    to be moved to the child cgroup itself to refresh css association.\n\n    This isn\u0027t a problem for cgroup_migrate_add_src() as each source\n    css_set still maps to single source and target cgroups; however,\n    cgroup_migrate_prepare_dst() is called once after all source css_sets\n    are added and thus might not have a single destination cgroup.  This\n    is currently worked around by specifying NULL for @dst_cgrp and using\n    the source\u0027s default cgroup as destination as the only multi-target\n    migration in use is self-targetting.  While this works, it\u0027s subtle\n    and clunky.\n\n    As all taget cgroups are already specified while preparing the source\n    css_sets, this clunkiness can easily be removed by recording the\n    target cgroup in each source css_set.  This patch adds\n    css_set-\u003emg_dst_cgrp which is recorded on cgroup_migrate_src() and\n    used by cgroup_migrate_prepare_dst().  This also makes migration code\n    ready for arbitrary multi-target migration.\n\n    Signed-off-by: Tejun Heo \u003ctj@kernel.org\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 6c415cca4afa1d48f972b5b4e1db426c783812e9\nAuthor: Tejun Heo \u003ctj@kernel.org\u003e\nDate:   Tue Mar 8 11:51:26 2016 -0500\n\n    cgroup: make cgroup[_taskset]_migrate() take cgroup_root instead of cgroup\n\n    On the default hierarchy, a migration can be multi-source and/or\n    multi-destination.  cgroup_taskest_migrate() used to incorrectly\n    assume single destination cgroup but the bug has been fixed by\n    1f7dd3e5a6e4 (\"cgroup: fix handling of multi-destination migration\n    from subtree_control enabling\").\n\n    Since the commit, @dst_cgrp to cgroup[_taskset]_migrate() is only used\n    to determine which subsystems are affected or which cgroup_root the\n    migration is taking place in.  As such, @dst_cgrp is misleading.  This\n    patch replaces @dst_cgrp with @root.\n\n    Signed-off-by: Tejun Heo \u003ctj@kernel.org\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit f9a273fd1c3ce5813c07494170af05e2f9107772\nAuthor: Tejun Heo \u003ctj@kernel.org\u003e\nDate:   Tue Mar 8 11:51:25 2016 -0500\n\n    cgroup: move migration destination verification out of cgroup_migrate_prepare_dst()\n\n    cgroup_migrate_prepare_dst() verifies whether the destination cgroup\n    is allowable; however, the test doesn\u0027t really belong there.  It\u0027s too\n    deep and common in the stack and as a result the test itself is gated\n    by another test.\n\n    Separate the test out into cgroup_may_migrate_to() and update\n    cgroup_attach_task() and cgroup_transfer_tasks() to perform the test\n    directly.  This doesn\u0027t cause any behavior differences.\n\n    Signed-off-by: Tejun Heo \u003ctj@kernel.org\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 553bf75db1692e5dfba70b8fdf0dd852ff7b8a09\nAuthor: Tejun Heo \u003ctj@kernel.org\u003e\nDate:   Tue Mar 8 11:51:25 2016 -0500\n\n    cgroup: fix incorrect destination cgroup in cgroup_update_dfl_csses()\n\n    cgroup_update_dfl_csses() should move each task in the subtree to\n    self; however, it was incorrectly calling cgroup_migrate_add_src()\n    with the root of the subtree as @dst_cgrp.  Fortunately,\n    cgroup_migrate_add_src() currently uses @dst_cgrp only to determine\n    the hierarchy and the bug doesn\u0027t cause any actual breakages.  Fix it.\n\n    Signed-off-by: Tejun Heo \u003ctj@kernel.org\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 79727d909f57eea7bba66d6512d1c3f6dd5bcc5a\nAuthor: Tejun Heo \u003ctj@kernel.org\u003e\nDate:   Thu Mar 3 09:58:01 2016 -0500\n\n    cgroup: update css iteration in cgroup_update_dfl_csses()\n\n    The existing sequences of operations ensure that the offlining csses\n    are drained before cgroup_update_dfl_csses(), so even though\n    cgroup_update_dfl_csses() uses css_for_each_descendant_pre() to walk\n    the target cgroups, it doesn\u0027t end up operating on dead cgroups.\n    Also, the function explicitly excludes the subtree root from\n    operation.\n\n    This is fragile and inconsistent with the rest of css update\n    operations.  This patch updates cgroup_update_dfl_csses() to use\n    cgroup_for_each_live_descendant_pre() instead and include the subtree\n    root.\n\n    Signed-off-by: Tejun Heo \u003ctj@kernel.org\u003e\n    Acked-by: Zefan Li \u003clizefan@huawei.com\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit ddf23b98ffcd402306cb9797a73efdce216d8c47\nAuthor: Tejun Heo \u003ctj@kernel.org\u003e\nDate:   Thu Mar 3 09:58:01 2016 -0500\n\n    cgroup: allocate 2x cgrp_cset_links when setting up a new root\n\n    During prep, cgroup_setup_root() allocates cgrp_cset_links matching\n    the number of existing css_sets to later link the new root.  This is\n    fine for now as the only operation which can happen inbetween is\n    rebind_subsystems() and rebinding of empty subsystems doesn\u0027t create\n    new css_sets.\n\n    However, while not yet allowed, with the recent reimplementation,\n    rebind_subsystems() can rebind subsystems with descendant csses and\n    thus can create new css_sets.  This patch makes cgroup_setup_root()\n    allocate 2x of the existing css_sets so that later use of live\n    subsystem rebinding doesn\u0027t blow up.\n\n    Signed-off-by: Tejun Heo \u003ctj@kernel.org\u003e\n    Acked-by: Zefan Li \u003clizefan@huawei.com\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit d1c5649b699a51aa7a13c0bbbafaecd0cebe2705\nAuthor: Tejun Heo \u003ctj@kernel.org\u003e\nDate:   Thu Mar 3 09:58:01 2016 -0500\n\n    cgroup: make cgroup_calc_subtree_ss_mask() take @this_ss_mask\n\n    cgroup_calc_subtree_ss_mask() currently takes @cgrp and\n    @subtree_control.  @cgrp is used for two purposes - to decide whether\n    it\u0027s for default hierarchy and the mask of available subsystems.  The\n    former doesn\u0027t matter as the results are the same regardless.  The\n    latter can be specified directly through a subsystem mask.\n\n    This patch makes cgroup_calc_subtree_ss_mask() perform the same\n    calculations for both default and legacy hierarchies and take\n    @this_ss_mask for available subsystems.  @cgrp is no longer used and\n    dropped.  This is to allow using the function in contexts where\n    available controllers can\u0027t be decided from the cgroup.\n\n    v2: cgroup_refres_subtree_ss_mask() is removed by a previous patch.\n        Updated accordingly.\n\n    Signed-off-by: Tejun Heo \u003ctj@kernel.org\u003e\n    Acked-by: Zefan Li \u003clizefan@huawei.com\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 879c688bf33b544222276a3ab085f007da05e239\nAuthor: Tejun Heo \u003ctj@kernel.org\u003e\nDate:   Thu Mar 3 09:58:01 2016 -0500\n\n    cgroup: reimplement rebind_subsystems() using cgroup_apply_control() and friends\n\n    rebind_subsystem() open codes quite a bit of css and interface file\n    manipulations.  It tries to be fail-safe but doesn\u0027t quite achieve it.\n    It can be greatly simplified by using the new css management helpers.\n    This patch reimplements rebind_subsytsems() using\n    cgroup_apply_control() and friends.\n\n    * The half-baked rollback on file creation failure is dropped.  It is\n      an extremely cold path, failure isn\u0027t critical, and, aside from\n      kernel bugs, the only reason it can fail is memory allocation\n      failure which pretty much doesn\u0027t happen for small allocations.\n\n    * As cgroup_apply_control_disable() is now used to clean up root\n      cgroup on rebind, make sure that it doesn\u0027t end up killing root\n      csses.\n\n    * All callers of rebind_subsystems() are updated to use\n      cgroup_lock_and_drain_offline() as the apply_control functions\n      require drained subtree.\n\n    * This leaves cgroup_refresh_subtree_ss_mask() without any user.\n      Removed.\n\n    * css_populate_dir() and css_clear_dir() no longer needs\n      @cgrp_override parameter.  Dropped.\n\n    * While at it, add WARN_ON() to rebind_subsystem() calls which are\n      expected to always succeed just in case.\n\n    While the rules visible to userland aren\u0027t changed, this\n    reimplementation not only simplifies rebind_subsystems() but also\n    allows it to disable and enable csses recursively.  This can be used\n    to implement more flexible rebinding.\n\n    Signed-off-by: Tejun Heo \u003ctj@kernel.org\u003e\n    Acked-by: Zefan Li \u003clizefan@huawei.com\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 843ec01cd0cb0d604ccef3a71b6c9db41a69c322\nAuthor: Tejun Heo \u003ctj@kernel.org\u003e\nDate:   Thu Mar 3 09:58:00 2016 -0500\n\n    cgroup: use cgroup_apply_enable_control() in cgroup creation path\n\n    cgroup_create() manually updates control masks and creates child csses\n    which cgroup_mkdir() then manually populates.  Both can be simplified\n    by using cgroup_apply_enable_control() and friends.  The only catch is\n    that it calls css_populate_dir() with NULL cgroup-\u003ekn during\n    cgroup_create().  This is worked around by making the function noop on\n    NULL kn.\n\n    Signed-off-by: Tejun Heo \u003ctj@kernel.org\u003e\n    Acked-by: Zefan Li \u003clizefan@huawei.com\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 860a4e3caf7c25c2f95311a2bc6126e6c3673ead\nAuthor: Tejun Heo \u003ctj@kernel.org\u003e\nDate:   Thu Mar 3 09:58:00 2016 -0500\n\n    cgroup: combine cgroup_mutex locking and offline css draining\n\n    cgroup_drain_offline() is used to wait for csses being offlined to\n    uninstall itself from cgroup-\u003esubsys[] array so that new csses can be\n    installed.  The function\u0027s only user, cgroup_subtree_control_write(),\n    calls it after performing some checks and restarts the whole process\n    via restart_syscall() if draining has to release cgroup_mutex to wait.\n\n    This can be simplified by draining before other synchronized\n    operations so that there\u0027s nothing to restart.  This patch converts\n    cgroup_drain_offline() to cgroup_lock_and_drain_offline() which\n    performs both locking and draining and updates cgroup_kn_lock_live()\n    use it instead of cgroup_mutex() if requested.  This combined locking\n    and draining operations are easier to use and less error-prone.\n\n    While at it, add WARNs in control_apply functions which triggers if\n    the subtree isn\u0027t properly drained.\n\n    Signed-off-by: Tejun Heo \u003ctj@kernel.org\u003e\n    Acked-by: Zefan Li \u003clizefan@huawei.com\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 64863fbc43e75c694a6d7baa2fddfa44472f6133\nAuthor: Tejun Heo \u003ctj@kernel.org\u003e\nDate:   Thu Mar 3 09:58:00 2016 -0500\n\n    cgroup: factor out cgroup_{apply|finalize}_control() from cgroup_subtree_control_write()\n\n    Factor out cgroup_{apply|finalize}_control() so that control mask\n    update can be done in several simple steps.  This patch doesn\u0027t\n    introduce behavior changes.\n\n    Signed-off-by: Tejun Heo \u003ctj@kernel.org\u003e\n    Acked-by: Zefan Li \u003clizefan@huawei.com\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 35246e6c4f8bfb235f6d844bfb3354d6fcc9ec92\nAuthor: Tejun Heo \u003ctj@kernel.org\u003e\nDate:   Thu Mar 3 09:57:59 2016 -0500\n\n    cgroup: introduce cgroup_{save|propagate|restore}_control()\n\n    While controllers are being enabled and disabled in\n    cgroup_subtree_control_write(), the original subsystem masks are\n    stashed in local variables so that they can be restored if the\n    operation fails in the middle.\n\n    This patch adds dedicated fields to struct cgroup to be used instead\n    of the local variables and implements functions to stash the current\n    values, propagate the changes and restore them recursively.  Combined\n    with the previous changes, this makes subsystem management operations\n    fully recursive and modularlized.  This will be used to expand cgroup\n    core functionalities.\n\n    While at it, remove now unused @css_enable and @css_disable from\n    cgroup_subtree_control_write().\n\n    Signed-off-by: Tejun Heo \u003ctj@kernel.org\u003e\n    Acked-by: Zefan Li \u003clizefan@huawei.com\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 89388014cdbce1c6780882cac41f3bf7d2b9a0a9\nAuthor: Tejun Heo \u003ctj@kernel.org\u003e\nDate:   Thu Mar 3 09:57:59 2016 -0500\n\n    cgroup: make cgroup_drain_offline() and cgroup_apply_control_{disable|enable}() recursive\n\n    The three factored out css management operations -\n    cgroup_drain_offline() and cgroup_apply_control_{disable|enable}() -\n    only depend on the current state of the target cgroups and idempotent\n    and thus can be easily made to operate on the subtree instead of the\n    immediate children.\n\n    This patch introduces the iterators which walk live subtree and\n    converts the three functions to operate on the subtree including self\n    instead of the children.  While this leads to spurious walking and be\n    slightly more expensive, it will allow them to be used for wider scope\n    of operations.\n\n    Note that cgroup_drain_offline() now tests for whether a css is dying\n    before trying to drain it.  This is to avoid trying to drain live\n    csses as there can be mix of live and dying csses in a subtree unlike\n    children of the same parent.\n\n    Signed-off-by: Tejun Heo \u003ctj@kernel.org\u003e\n    Acked-by: Zefan Li \u003clizefan@huawei.com\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit d6a5af7d1bbe54165457091b329d4b6118732d41\nAuthor: Tejun Heo \u003ctj@kernel.org\u003e\nDate:   Thu Mar 3 09:57:59 2016 -0500\n\n    cgroup: factor out cgroup_apply_control_enable() from cgroup_subtree_control_write()\n\n    Factor out css enabling and showing into cgroup_apply_control_enable().\n\n    * Nest subsystem walk inside child walk.  The child walk will later be\n      converted to subtree walk which is a bit more expensive.\n\n    * Instead of operating on the differential masks @css_enable, simply\n      enable or show csses according to the current cgroup_control() and\n      cgroup_ss_mask().  This leads to the same result and is simpler and\n      more robust.\n\n    Signed-off-by: Tejun Heo \u003ctj@kernel.org\u003e\n    Acked-by: Zefan Li \u003clizefan@huawei.com\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 6a92fb91f8fb9e7d5bd6174dd11ab6fa611fb4ab\nAuthor: Tejun Heo \u003ctj@kernel.org\u003e\nDate:   Thu Mar 3 09:57:59 2016 -0500\n\n    cgroup: factor out cgroup_apply_control_disable() from cgroup_subtree_control_write()\n\n    Factor out css disabling and hiding into cgroup_apply_control_disable().\n\n    * Nest subsystem walk inside child walk.  The child walk will later be\n      converted to subtree walk which is a bit more expensive.\n\n    * Instead of operating on the differential masks @css_enable and\n      @css_disable, simply disable or hide csses according to the current\n      cgroup_control() and cgroup_ss_mask().  This leads to the same\n      result and is simpler and more robust.\n\n    * This allows error handling path to share the same code.\n\n    Signed-off-by: Tejun Heo \u003ctj@kernel.org\u003e\n    Acked-by: Zefan Li \u003clizefan@huawei.com\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 86f2a0316967cb9401593ef03ccc2e8fc9f67710\nAuthor: Tejun Heo \u003ctj@kernel.org\u003e\nDate:   Thu Mar 3 09:57:59 2016 -0500\n\n    cgroup: factor out cgroup_drain_offline() from cgroup_subtree_control_write()\n\n    Factor out async css offline draining into cgroup_drain_offline().\n\n    * Nest subsystem walk inside child walk.  The child walk will later be\n      converted to subtree walk which is a bit more expensive.\n\n    * Relocate the draining above subsystem mask preparation, which\n      doesn\u0027t create any behavior differences but helps further\n      refactoring.\n\n    Signed-off-by: Tejun Heo \u003ctj@kernel.org\u003e\n    Acked-by: Zefan Li \u003clizefan@huawei.com\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit ad4f46e140df34c008384e7e59562a8a64461f18\nAuthor: Tejun Heo \u003ctj@kernel.org\u003e\nDate:   Thu Mar 3 09:57:58 2016 -0500\n\n    cgroup: introduce cgroup_control() and cgroup_ss_mask()\n\n    When a controller is enabled and visible on a non-root cgroup is\n    determined by subtree_control and subtree_ss_mask of the parent\n    cgroup.  For a root cgroup, by the type of the hierarchy and which\n    controllers are attached to it.  Deciding the above on each usage is\n    fragile and unnecessarily complicates the users.\n\n    This patch introduces cgroup_control() and cgroup_ss_mask() which\n    calculate and return the [visibly] enabled subsyste mask for the\n    specified cgroup and conver the existing usages.\n\n    * cgroup_e_css() is restructured for simplicity.\n\n    * cgroup_calc_subtree_ss_mask() and cgroup_subtree_control_write() no\n      longer need to distinguish root and non-root cases.\n\n    * With cgroup_control(), cgroup_controllers_show() can now handle both\n      root and non-root cases.  cgroup_root_controllers_show() is removed.\n\n    v2: cgroup_control() updated to yield the correct result on v1\n        hierarchies too.  cgroup_subtree_control_write() converted.\n\n    Signed-off-by: Tejun Heo \u003ctj@kernel.org\u003e\n    Acked-by: Zefan Li \u003clizefan@huawei.com\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 5ce230f708b4a27a09b0cbeb8a8572ead57802f8\nAuthor: Tejun Heo \u003ctj@kernel.org\u003e\nDate:   Thu Mar 3 09:57:58 2016 -0500\n\n    cgroup: factor out cgroup_create() out of cgroup_mkdir()\n\n    We\u0027re in the process of refactoring cgroup and css management paths to\n    separate them out to eventually allow cgroups which aren\u0027t visible\n    through cgroup fs.  This patch factors out cgroup_create() out of\n    cgroup_mkdir().  cgroup_create() contains all internal object creation\n    and initialization.  cgroup_mkdir() uses cgroup_create() to create the\n    internal cgroup and adds interface directory and file creation.\n\n    This patch doesn\u0027t cause any behavior differences.\n\n    Signed-off-by: Tejun Heo \u003ctj@kernel.org\u003e\n    Acked-by: Zefan Li \u003clizefan@huawei.com\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit e672d8a005bf3e08e4931841f77483ed9966a778\nAuthor: Tejun Heo \u003ctj@kernel.org\u003e\nDate:   Thu Mar 3 09:57:58 2016 -0500\n\n    cgroup: reorder operations in cgroup_mkdir()\n\n    Currently, operations to initialize internal objects and create\n    interface directory and files are intermixed in cgroup_mkdir().  We\u0027re\n    in the process of refactoring cgroup and css management paths to\n    separate them out to eventually allow cgroups which aren\u0027t visible\n    through cgroup fs.\n\n    This patch reorders operations inside cgroup_mkdir() so that interface\n    directory and file handling comes after internal object\n    initialization.  This will enable further refactoring.\n\n    Signed-off-by: Tejun Heo \u003ctj@kernel.org\u003e\n    Acked-by: Zefan Li \u003clizefan@huawei.com\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 0676b7a4b1fca64e34a897a03566c011d2dc3fcc\nAuthor: Tejun Heo \u003ctj@kernel.org\u003e\nDate:   Thu Mar 3 09:57:58 2016 -0500\n\n    cgroup: explicitly track whether a cgroup_subsys_state is visible to userland\n\n    Currently, whether a css (cgroup_subsys_state) has its interface files\n    created is not tracked and assumed to change together with the owning\n    cgroup\u0027s lifecycle.  cgroup directory and interface creation is being\n    separated out from internal object creation to help refactoring and\n    eventually allow cgroups which are not visible through cgroupfs.\n\n    This patch adds CSS_VISIBLE to track whether a css has its interface\n    files created and perform management operations only when necessary\n    which helps decoupling interface file handling from internal object\n    lifecycle.  After this patch, all css interface file management\n    functions can be called regardless of the current state and will\n    achieve the expected result.\n\n    Signed-off-by: Tejun Heo \u003ctj@kernel.org\u003e\n    Acked-by: Zefan Li \u003clizefan@huawei.com\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 2befc2460c74038c0e87483872031bb8df4cbce5\nAuthor: Tejun Heo \u003ctj@kernel.org\u003e\nDate:   Thu Mar 3 09:57:58 2016 -0500\n\n    cgroup: separate out interface file creation from css creation\n\n    Currently, interface files are created when a css is created depending\n    on whether @visible is set.  This patch separates out the two into\n    separate steps to help code refactoring and eventually allow cgroups\n    which aren\u0027t visible through cgroup fs.\n\n    Move css_populate_dir() out of create_css() and drop @visible.  While\n    at it, rename the function to css_create() for consistency.\n\n    Signed-off-by: Tejun Heo \u003ctj@kernel.org\u003e\n    Acked-by: Zefan Li \u003clizefan@huawei.com\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 935698acba9d5f2b555ad1e15fec1248b16c39a0\nAuthor: Tejun Heo \u003ctj@kernel.org\u003e\nDate:   Thu Mar 3 09:57:57 2016 -0500\n\n    cgroup: suppress spurious de-populated events\n\n    During task migration, tasks may transfer between two css_sets which\n    are associated with the same cgroup.  If those tasks are the only\n    tasks in the cgroup, this currently triggers a spurious de-populated\n    event on the cgroup.\n\n    Fix it by bumping up populated count before bumping it down during\n    migration to ensure that it doesn\u0027t reach zero spuriously.\n\n    Signed-off-by: Tejun Heo \u003ctj@kernel.org\u003e\n    Acked-by: Zefan Li \u003clizefan@huawei.com\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 1db7f37aac8a2a191202ff9e4717e3f2945012a8\nAuthor: Tejun Heo \u003ctj@kernel.org\u003e\nDate:   Thu Mar 3 09:57:57 2016 -0500\n\n    cgroup: re-hash init_css_set after subsystems are initialized\n\n    css_sets are hashed by their subsys[] contents and in cgroup_init()\n    init_css_set is hashed early, before subsystem inits, when all entries\n    in its subsys[] are NULL, so that cgroup_dfl_root initialization can\n    find and link to it.  As subsystems are initialized,\n    init_css_set.subsys[] is filled up but the hashing is never updated\n    making init_css_set hashed in the wrong place.  While incorrect, this\n    doesn\u0027t cause a critical failure as css_set management code would\n    create an identical css_set dynamically.\n\n    Fix it by rehashing init_css_set after subsystems are initialized.\n    While at it, drop unnecessary @key local variable.\n\n    Signed-off-by: Tejun Heo \u003ctj@kernel.org\u003e\n    Acked-by: Zefan Li \u003clizefan@huawei.com\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit dc818c7fea03fea726845aa5d25afac6e8ed3020\nAuthor: Vladimir Davydov \u003cvdavydov@virtuozzo.com\u003e\nDate:   Tue Mar 1 19:56:30 2016 +0300\n\n    cgroup: reset css on destruction\n\n    An associated css can be around for quite a while after a cgroup\n    directory has been removed. In general, it makes sense to reset it to\n    defaults so as not to worry about any remnants. For instance, memory\n    cgroup needs to reset memory.low, otherwise pages charged to a dead\n    cgroup might never get reclaimed. There\u0027s -\u003ecss_reset callback, which\n    would fit perfectly for the purpose. Currently, it\u0027s only called when a\n    subsystem is disabled in the unified hierarchy and there are other\n    subsystems dependant on it. Let\u0027s call it on css destruction as well.\n\n    Suggested-by: Johannes Weiner \u003channes@cmpxchg.org\u003e\n    Signed-off-by: Vladimir Davydov \u003cvdavydov@virtuozzo.com\u003e\n    Signed-off-by: Tejun Heo \u003ctj@kernel.org\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 4e74adb6b4bad377d78f350044ff81bf1f3923ba\nAuthor: Tejun Heo \u003ctj@kernel.org\u003e\nDate:   Sun Feb 28 08:59:33 2016 -0500\n\n    cgroup: fix and restructure error handling in copy_cgroup_ns()\n\n    copy_cgroup_ns()\u0027s error handling was broken and the attempt to fix it\n    d22025570e2e (\"cgroup: fix alloc_cgroup_ns() error handling in\n    copy_cgroup_ns()\") was broken too in that it ended up trying an\n    ERR_PTR() value.\n\n    There\u0027s only one place where copy_cgroup_ns() needs to perform cleanup\n    after failure.  Simplify and fix the error handling by removing the\n    goto\u0027s.\n\n    Signed-off-by: Tejun Heo \u003ctj@kernel.org\u003e\n    Reported-by: Dan Carpenter \u003cdan.carpenter@oracle.com\u003e\n    Acked-by: Serge E. Hallyn \u003cserge.hallyn@ubuntu.com\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 131290672d7c682ec38ed7e3e3359e396e153b1c\nAuthor: Xiubo Li \u003clixiubo@cmss.chinamobile.com\u003e\nDate:   Fri Feb 26 13:07:38 2016 +0800\n\n    cgroup: fix a mistake in warning message\n\n    There is a mistake about the print format name:id \u003c--\u003e %d:%s, which\n    the name is \u0027char *\u0027 type and id is \u0027int\u0027 type.  Change \"name:id\" to\n    \"id:name\" instead to be consistent with \"cgroup_subsys %d:%s\".\n\n    Signed-off-by: Xiubo Li \u003clixiubo@cmss.chinamobile.com\u003e\n    Acked-by: Zefan Li \u003clizefan@huawei.com\u003e\n    Signed-off-by: Tejun Heo \u003ctj@kernel.org\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 9a2d67655e17acd0f9748f8b4b3a25932b1bca97\nAuthor: Tejun Heo \u003ctj@kernel.org\u003e\nDate:   Tue Feb 23 10:00:51 2016 -0500\n\n    cgroup: use -\u003esubtree_control when testing no internal process rule\n\n    No internal process rule is enforced by cgroup_migrate_prepare_dst()\n    during process migration.  It tests whether the target cgroup\u0027s\n    -\u003echild_subsys_mask is zero which is different from \"subtree_control\"\n    write path which tests -\u003esubtree_control.  This hasn\u0027t mattered\n    because up until now, both -\u003echild_subsys_mask and -\u003esubtree_control\n    are zero or non-zero at the same time.  However, with the planned\n    addition of implicit controllers, this will no longer be true.\n\n    This patch prepares for the change by making\n    cgorup_migrate_prepare_dst() test -\u003esubtree_control instead.\n\n    Signed-off-by: Tejun Heo \u003ctj@kernel.org\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit d289cc8ebc7a51cfe3a99c7bdd74950d3cb37809\nAuthor: Tejun Heo \u003ctj@kernel.org\u003e\nDate:   Tue Feb 23 10:00:51 2016 -0500\n\n    cgroup: make css_tryget_online_from_dir() also recognize cgroup2 fs\n\n    The function currently returns -EBADF for a directory on the default\n    hierarchy.  Make it also recognize cgroup2_fs_type.  This will be used\n    for perf_event cgroup2 support.\n\n    Signed-off-by: Tejun Heo \u003ctj@kernel.org\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit c413d3cc43a2778b8ebfbc71d9522c69a9ea443c\nAuthor: Tejun Heo \u003ctj@kernel.org\u003e\nDate:   Tue Feb 23 10:00:50 2016 -0500\n\n    cgroup: s/cgrp_dfl_root_/cgrp_dfl_/\n\n    These var names are unnecessarily unwiedly and another similar\n    variable will be added.  Let\u0027s shorten them.\n\n    Signed-off-by: Tejun Heo \u003ctj@kernel.org\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 8e15d41dcac42fa8ff1ee04fe7306759d5a8f94f\nAuthor: Tejun Heo \u003ctj@kernel.org\u003e\nDate:   Mon Feb 22 22:25:47 2016 -0500\n\n    cgroup: make cgroup subsystem masks u16\n\n    After the recent do_each_subsys_mask() conversion, there\u0027s no reason\n    to use ulong for subsystem masks.  We\u0027ll be adding more subsystem\n    masks to persistent data structures, let\u0027s reduce its size to u16\n    which should be enough for now and the foreseeable future.\n\n    This doesn\u0027t create any noticeable behavior differences.\n\n    v2: Johannes spotted that the initial patch missed cgroup_no_v1_mask.\n        Converted.\n\n    Signed-off-by: Tejun Heo \u003ctj@kernel.org\u003e\n    Acked-by: Johannes Weiner \u003channes@cmpxchg.org\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit eb70a3cc9b33ff80f275334d0f4d3f167ac4f24a\nAuthor: Tejun Heo \u003ctj@kernel.org\u003e\nDate:   Mon Feb 22 22:25:46 2016 -0500\n\n    cgroup: use do_each_subsys_mask() where applicable\n\n    There are several places in cgroup_subtree_control_write() which can\n    use do_each_subsys_mask() instead of manual mask testing.  Use it.\n\n    No functional changes.\n\n    Signed-off-by: Tejun Heo \u003ctj@kernel.org\u003e\n    Acked-by: Johannes Weiner \u003channes@cmpxchg.org\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 61a6a4a003e763c5b442af6e1a7e7a6f7192210c\nAuthor: Tejun Heo \u003ctj@kernel.org\u003e\nDate:   Mon Feb 22 22:25:46 2016 -0500\n\n    cgroup: convert for_each_subsys_which() to do-while style\n\n    for_each_subsys_which() allows iterating subsystems specified in a\n    subsystem bitmask; unfortunately, it requires the mask to be an\n    unsigned long l-value which can be inconvenient and makes it awkward\n    to use a smaller type for subsystem masks.\n\n    This patch converts for_each_subsy_which() to do-while style which\n    allows it to drop the l-value requirement.  The new iterator is named\n    do_each_subsys_mask() / while_each_subsys_mask().\n\n    Signed-off-by: Tejun Heo \u003ctj@kernel.org\u003e\n    Cc: Aleksa Sarai \u003ccyphar@cyphar.com\u003e\n    Acked-by: Johannes Weiner \u003channes@cmpxchg.org\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit b375f813de76b62b9d5bd1e146e998b4a1332a25\nAuthor: Tejun Heo \u003ctj@kernel.org\u003e\nDate:   Mon Feb 22 22:25:46 2016 -0500\n\n    cgroup: s/child_subsys_mask/subtree_ss_mask/\n\n    For consistency with cgroup-\u003esubtree_control.\n\n    * cgroup-\u003echild_subsys_mask -\u003e cgroup-\u003esubtree_ss_mask\n    * cgroup_calc_child_subsys_mask() -\u003e cgroup_calc_subtree_ss_mask()\n    * cgroup_refresh_child_subsys_mask() -\u003e cgroup_refresh_subtree_ss_mask()\n\n    No functional changes.\n\n    Signed-off-by: Tejun Heo \u003ctj@kernel.org\u003e\n    Acked-by: Johannes Weiner \u003channes@cmpxchg.org\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit a761cb8173b56e7b3bbe97db8ecea2fcdf9fb417\nAuthor: Tejun Heo \u003ctj@kernel.org\u003e\nDate:   Mon Feb 22 22:25:46 2016 -0500\n\n    Revert \"cgroup: add cgroup_subsys-\u003ecss_e_css_changed()\"\n\n    This reverts commit 56c807ba4e91f0980567b6a69de239677879b17f.\n\n    cgroup_subsys-\u003ecss_e_css_changed() was supposed to be used by cgroup\n    writeback support; however, the change to per-inode cgroup association\n    made it unnecessary and the callback doesn\u0027t have any user.  Remove\n    it.\n\n    Signed-off-by: Tejun Heo \u003ctj@kernel.org\u003e\n    Acked-by: Johannes Weiner \u003channes@cmpxchg.org\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 0b386a52c6c7c7b6950a71aa397ca9e49d0febc4\nAuthor: Tejun Heo \u003ctj@kernel.org\u003e\nDate:   Mon Feb 22 22:25:45 2016 -0500\n\n    cgroup: fix error return value of cgroup_addrm_files()\n\n    cgroup_addrm_files() incorrectly returned 0 after add failure.  Fix\n    it.\n\n    Signed-off-by: Tejun Heo \u003ctj@kernel.org\u003e\n    Acked-by: Johannes Weiner \u003channes@cmpxchg.org\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit de086d32c1071a1543e0be849e0db796de77308c\nAuthor: Tejun Heo \u003ctj@kernel.org\u003e\nDate:   Thu Feb 18 11:44:24 2016 -0500\n\n    cgroup: fix alloc_cgroup_ns() error handling in copy_cgroup_ns()\n\n    alloc_cgroup_ns() returns an ERR_PTR value on error but\n    copy_cgroup_ns() was checking for NULL for error.  Fix it.\n\n    Signed-off-by: Tejun Heo \u003ctj@kernel.org\u003e\n    Reported-by: Dan Carpenter \u003cdan.carpenter@oracle.com\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit b099900c0607c7df697cca646a89f27a26b8df2b\nAuthor: Serge Hallyn \u003cserge.hallyn@ubuntu.com\u003e\nDate:   Fri Jan 29 02:54:11 2016 -0600\n\n    Add FS_USERNS_FLAG to cgroup fs\n\n    allowing root in a non-init user namespace to mount it.  This should\n    now be safe, because\n\n    1. non-init-root cannot mount a previously unbound subsystem\n    2. the task doing the mount must be privileged with respect to the\n       user namespace owning the cgroup namespace\n    3. the mounted subsystem will have its current cgroup as the root dentry.\n       the permissions will be unchanged, so tasks will receive no new\n       privilege over the cgroups which they did not have on the original\n       mounts.\n\n    Signed-off-by: Serge Hallyn \u003cserge.hallyn@canonical.com\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit d29afb11b84109e3dc00823e8f7d1de18c1ec843\nAuthor: Serge Hallyn \u003cserge.hallyn@ubuntu.com\u003e\nDate:   Fri Jan 29 02:54:09 2016 -0600\n\n    cgroup: mount cgroupns-root when inside non-init cgroupns\n\n    This patch enables cgroup mounting inside userns when a process\n    as appropriate privileges. The cgroup filesystem mounted is\n    rooted at the cgroupns-root. Thus, in a container-setup, only\n    the hierarchy under the cgroupns-root is exposed inside the container.\n    This allows container management tools to run inside the containers\n    without depending on any global state.\n\n    Signed-off-by: Serge Hallyn \u003cserge.hallyn@canonical.com\u003e\n    Signed-off-by: Tejun Heo \u003ctj@kernel.org\u003e\n    Signed-off-by: Tejun Heo \u003ctj@kernel.org\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n    Change-Id: Ic4361ad382420757ac21fef7f5ac41d54cd69b6b\n\ncommit 46a283f4b2bd09e50b2f725a272c77121db84e15\nAuthor: Aditya Kali \u003cadityakali@google.com\u003e\nDate:   Fri Jan 29 02:54:07 2016 -0600\n\n    cgroup: cgroup namespace setns support\n\n    setns on a cgroup namespace is allowed only if\n    task has CAP_SYS_ADMIN in its current user-namespace and\n    over the user-namespace associated with target cgroupns.\n    No implicit cgroup changes happen with attaching to another\n    cgroupns. It is expected that the somone moves the attaching\n    process under the target cgroupns-root.\n\n    Signed-off-by: Aditya Kali \u003cadityakali@google.com\u003e\n    Signed-off-by: Serge E. Hallyn \u003cserge.hallyn@canonical.com\u003e\n    Signed-off-by: Tejun Heo \u003ctj@kernel.org\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 19bdd93e9d2544b17e173d81cff6e26e6a69f812\nAuthor: Aditya Kali \u003cadityakali@google.com\u003e\nDate:   Fri Jan 29 02:54:06 2016 -0600\n\n    cgroup: introduce cgroup namespaces\n\n    Introduce the ability to create new cgroup namespace. The newly created\n    cgroup namespace remembers the cgroup of the process at the point\n    of creation of the cgroup namespace (referred as cgroupns-root).\n    The main purpose of cgroup namespace is to virtualize the contents\n    of /proc/self/cgroup file. Processes inside a cgroup namespace\n    are only able to see paths relative to their namespace root\n    (unless they are moved outside of their cgroupns-root, at which point\n     they will see a relative path from their cgroupns-root).\n    For a correctly setup container this enables container-tools\n    (like libcontainer, lxc, lmctfy, etc.) to create completely virtualized\n    containers without leaking system level cgroup hierarchy to the task.\n    This patch only implements the \u0027unshare\u0027 part of the cgroupns.\n\n    Signed-off-by: Aditya Kali \u003cadityakali@google.com\u003e\n    Signed-off-by: Serge Hallyn \u003cserge.hallyn@canonical.com\u003e\n    Signed-off-by: Tejun Heo \u003ctj@kernel.org\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n    Change-Id: I41c6f048567d7ab086467c6a230780c1858b315d\n\ncommit fd51a4a67a22dfb7c777dc8f1e2b0d09c8e82dcc\nAuthor: Johannes Weiner \u003channes@cmpxchg.org\u003e\nDate:   Thu Feb 11 13:34:49 2016 -0500\n\n    cgroup: provide cgroup_nov1\u003d to disable controllers in v1 mounts\n\n    Testing cgroup2 can be painful with system software automatically\n    mounting and populating all cgroup controllers in v1 mode. Sometimes\n    they can be unmounted from rc.local, sometimes even that is too late.\n\n    Provide a commandline option to disable certain controllers in v1\n    mounts, so that they remain available for cgroup2 mounts.\n\n    Example use:\n\n    cgroup_no_v1\u003dmemory,cpu\n    cgroup_no_v1\u003dall\n\n    Disabling will be confirmed at boot-time as such:\n\n    [    0.013770] Disabling cpu control group subsystem in v1 mounts\n    [    0.016004] Disabling memory control group subsystem in v1 mounts\n\n    Signed-off-by: Johannes Weiner \u003channes@cmpxchg.org\u003e\n    Signed-off-by: Tejun Heo \u003ctj@kernel.org\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 6a3fb344b8368e662ddc87bcb36b43175a8ba55e\nAuthor: Tejun Heo \u003ctj@kernel.org\u003e\nDate:   Tue Dec 29 14:53:56 2015 -0500\n\n    cgroup: demote subsystem init messages to KERN_DEBUG\n\n    These are noisy during boot and not all that interesting.\n\n    Signed-off-by: Tejun Heo \u003ctj@kernel.org\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 0fbac5eaf0a2d1a44a5e1b6c2bb1c14d07c7f348\nAuthor: Rami Rosen \u003crami.rosen@intel.com\u003e\nDate:   Sat Jan 9 23:33:06 2016 +0200\n\n    cgroup: fix a typo.\n\n    This patch fixes a typo in a comment in cgroup.c.\n\n    Signed-off-by: Rami Rosen \u003crami.rosen@intel.com\u003e\n    Signed-off-by: Tejun Heo \u003ctj@kernel.org\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 7540f1f6e01d59f71b092c2915abfceccf24acff\nAuthor: Tejun Heo \u003ctj@kernel.org\u003e\nDate:   Mon Dec 14 11:24:06 2015 -0500\n\n    net, cgroup: cgroup_sk_updat_lock was missing initializer\n\n    bd1060a1d671 (\"sock, cgroup: add sock-\u003esk_cgroup\") added global\n    spinlock cgroup_sk_update_lock but erroneously skipped initializer\n    leading to uninitialized spinlock warning.  Fix it by using\n    DEFINE_SPINLOCK().\n\n    Signed-off-by: Tejun Heo \u003ctj@kernel.org\u003e\n    Reported-by: Dexuan Cui \u003cdecui@microsoft.com\u003e\n    Fixes: bd1060a1d671 (\"sock, cgroup: add sock-\u003esk_cgroup\")\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 154e9825539dc382cc226bfd1ee2387f1712ccfc\nAuthor: Tejun Heo \u003ctj@kernel.org\u003e\nDate:   Mon Dec 7 17:38:53 2015 -0500\n\n    sock, cgroup: add sock-\u003esk_cgroup\n\n    In cgroup v1, dealing with cgroup membership was difficult because the\n    number of membership associations was unbound.  As a result, cgroup v1\n    grew several controllers whose primary purpose is either tagging\n    membership or pull in configuration knobs from other subsystems so\n    that cgroup membership test can be avoided.\n\n    net_cls and net_prio controllers are examples of the latter.  They\n    allow configuring network-specific attributes from cgroup side so that\n    network subsystem can avoid testing cgroup membership; unfortunately,\n    these are not only cumbersome but also problematic.\n\n    Both net_cls and net_prio aren\u0027t properly hierarchical.  Both inherit\n    configuration from the parent on creation but there\u0027s no interaction\n    afterwards.  An ancestor doesn\u0027t restrict the behavior in its subtree\n    in anyway and configuration changes aren\u0027t propagated downwards.\n    Especially when combined with cgroup delegation, this is problematic\n    because delegatees can mess up whatever network configuration\n    implemented at the system level.  net_prio would allow the delegatees\n    to set whatever priority value regardless of CAP_NET_ADMIN and net_cls\n    the same for classid.\n\n    While it is possible to solve these issues from controller side by\n    implementing hierarchical allowable ranges in both controllers, it\n    would involve quite a bit of complexity in the controllers and further\n    obfuscate network configuration as it becomes even more difficult to\n    tell what\u0027s actually being configured looking from the network side.\n    While not much can be done for v1 at this point, as membership\n    handling is sane on cgroup v2, it\u0027d be better to make cgroup matching\n    behave like other network matches and classifiers than introducing\n    further complications.\n\n    In preparation, this patch updates sock-\u003esk_cgrp_data handling so that\n    it points to the v2 cgroup that sock was created in until either\n    net_prio or net_cls is used.  Once either of the two is used,\n    sock-\u003esk_cgrp_data reverts to its previous role of carrying prioidx\n    and classid.  This is to avoid adding yet another cgroup related field\n    to struct sock.\n\n    As the mode switching can happen at most once per boot, the switching\n    mechanism is aimed at lowering hot path overhead.  It may leak a\n    finite, likely small, number of cgroup refs and report spurious\n    prioidx or classid on switching; however, dynamic updates of prioidx\n    and classid have always been racy and lossy - socks between creation\n    and fd installation are never updated, config changes don\u0027t update\n    existing sockets at all, and prioidx may index with dead and recycled\n    cgroup IDs.  Non-critical inaccuracies from small race windows won\u0027t\n    make any noticeable difference.\n\n    This patch doesn\u0027t make use of the pointer yet.  The following patch\n    will implement netfilter match for cgroup2 membership.\n\n    v2: Use sock_cgroup_data to avoid inflating struct sock w/ another\n        cgroup specific field.\n\n    v3: Add comments explaining why sock_data_prioidx() and\n        sock_data_classid() use different fallback values.\n\n    Signed-off-by: Tejun Heo \u003ctj@kernel.org\u003e\n    Cc: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Cc: Daniel Wagner \u003cdaniel.wagner@bmw-carit.de\u003e\n    CC: Neil Horman \u003cnhorman@tuxdriver.com\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 364d45aadac7bf888a4c6253315fc8b5b5c17089\nAuthor: Tejun Heo \u003ctj@kernel.org\u003e\nDate:   Mon Dec 7 17:38:52 2015 -0500\n\n    net: wrap sock-\u003esk_cgrp_prioidx and -\u003esk_classid inside a struct\n\n    Introduce sock-\u003esk_cgrp_data which is a struct sock_cgroup_data.\n    -\u003esk_cgroup_prioidx and -\u003esk_classid are moved into it.  The struct\n    and its accessors are defined in cgroup-defs.h.  This is to prepare\n    for overloading the fields with a cgroup pointer.\n\n    This patch mostly performs equivalent conversions but the followings\n    are noteworthy.\n\n    * Equality test before updating classid is removed from\n      sock_update_classid().  This shouldn\u0027t make any noticeable\n      difference and a similar test will be implemented on the helper side\n      later.\n\n    * sock_update_netprioidx() now takes struct sock_cgroup_data and can\n      be moved to netprio_cgroup.h without causing include dependency\n      loop.  Moved.\n\n    * The dummy version of sock_update_netprioidx() converted to a static\n      inline function while at it.\n\n    Signed-off-by: Tejun Heo \u003ctj@kernel.org\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit ba95ea9a2ea8346a2dc434fedf7560c5e792c18e\nAuthor: Tejun Heo \u003ctj@kernel.org\u003e\nDate:   Mon Dec 7 17:38:51 2015 -0500\n\n    netprio_cgroup: limit the maximum css-\u003eid to USHRT_MAX\n\n    netprio builds per-netdev contiguous priomap array which is indexed by\n    css-\u003eid.  The array is allocated using kzalloc() effectively limiting\n    the maximum ID supported to some thousand range.  This patch caps the\n    maximum supported css-\u003eid to USHRT_MAX which should be way above what\n    is actually useable.\n\n    This allows reducing sock-\u003esk_cgrp_prioidx to u16 from u32.  The freed\n    up part will be used to overload the cgroup related fields.\n    sock-\u003esk_cgrp_prioidx\u0027s position is swapped with sk_mark so that the\n    two cgroup related fields are adjacent.\n\n    Signed-off-by: Tejun Heo \u003ctj@kernel.org\u003e\n    Acked-by: Daniel Wagner \u003cdaniel.wagner@bmw-carit.de\u003e\n    Cc: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    CC: Neil Horman \u003cnhorman@tuxdriver.com\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 388d0685a2af4859b75fa8b1d41c4fbb10b54a4f\nAuthor: Oleg Nesterov \u003coleg@redhat.com\u003e\nDate:   Thu Dec 3 10:24:08 2015 -0500\n\n    cgroup: kill cgrp_ss_priv[CGROUP_CANFORK_COUNT] and friends\n\n    Now that nobody use the \"priv\" arg passed to can_fork/cancel_fork/fork we can\n    kill CGROUP_CANFORK_COUNT/SUBSYS_TAG/etc and cgrp_ss_priv[] in copy_process().\n\n    Signed-off-by: Oleg Nesterov \u003coleg@redhat.com\u003e\n    Signed-off-by: Tejun Heo \u003ctj@kernel.org\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n    Change-Id: I153eb067c3378de42ccfd4bf114763032bb260ec\n\ncommit 20868494e53f59fc11427a43b6b75a6938014c24\nAuthor: Tejun Heo \u003ctj@kernel.org\u003e\nDate:   Fri Nov 20 15:55:52 2015 -0500\n\n    cgroup: implement cgroup_get_from_path() and expose cgroup_put()\n\n    Implement cgroup_get_from_path() using kernfs_walk_and_get() which\n    obtains a default hierarchy cgroup from its path.  This will be used\n    to allow cgroup path based matching from outside cgroup proper -\n    e.g. networking and perf.\n\n    v2: Add EXPORT_SYMBOL_GPL(cgroup_get_from_path).\n\n    Signed-off-by: Tejun Heo \u003ctj@kernel.org\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 6137ad87a1b3fde20e32cdb34b1e7fb12f2ace37\nAuthor: Tejun Heo \u003ctj@kernel.org\u003e\nDate:   Fri Nov 20 15:55:52 2015 -0500\n\n    cgroup: record ancestor IDs and reimplement cgroup_is_descendant() using it\n\n    cgroup_is_descendant() currently walks up the hierarchy and compares\n    each ancestor to the cgroup in question.  While enough for cgroup core\n    usages, this can\u0027t be used in hot paths to test cgroup membership.\n    This patch adds cgroup-\u003eancestor_ids[] which records the IDs of all\n    ancestors including self and cgroup-\u003elevel for the nesting level.\n\n    This allows testing whether a given cgroup is a descendant of another\n    in three finite steps - testing whether the two belong to the same\n    hierarchy, whether the descendant candidate is at the same or a higher\n    level than the ancestor and comparing the recorded ancestor_id at the\n    matching level.  cgroup_is_descendant() is accordingly reimplmented\n    and made inline.\n\n    Signed-off-by: Tejun Heo \u003ctj@kernel.org\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 5e85046866cd4481ba302d8c255fdab0bc1d794f\nAuthor: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\nDate:   Mon May 8 00:04:09 2017 +0200\n\n    bpf: don\u0027t let ldimm64 leak map addresses on unprivileged\n\n    [ Upstream commit 0d0e57697f162da4aa218b5feafe614fb666db07 ]\n\n    The patch fixes two things at once:\n\n    1) It checks the env-\u003eallow_ptr_leaks and only prints the map address to\n       the log if we have the privileges to do so, otherwise it just dumps 0\n       as we would when kptr_restrict is enabled on %pK. Given the latter is\n       off by default and not every distro sets it, I don\u0027t want to rely on\n       this, hence the 0 by default for unprivileged.\n\n    2) Printing of ldimm64 in the verifier log is currently broken in that\n       we don\u0027t print the full immediate, but only the 32 bit part of the\n       first insn part for ldimm64. Thus, fix this up as well; it\u0027s okay to\n       access, since we verified all ldimm64 earlier already (including just\n       constants) through replace_map_fd_with_map_ptr().\n\n    Fixes: 1be7f75d1668 (\"bpf: enable non-root eBPF programs\")\n    Fixes: cbd357008604 (\"bpf: verifier (add ability to receive verification log)\")\n    Reported-by: Jann Horn \u003cjannh@google.com\u003e\n    Signed-off-by: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Acked-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Greg Kroah-Hartman \u003cgregkh@linuxfoundation.org\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit add65ebe457537b6247876a624adaf575debf976\nAuthor: Yonghong Song \u003cyhs@fb.com\u003e\nDate:   Sat Apr 29 22:52:42 2017 -0700\n\n    bpf: enhance verifier to understand stack pointer arithmetic\n\n    [ Upstream commit 332270fdc8b6fba07d059a9ad44df9e1a2ad4529 ]\n\n    llvm 4.0 and above generates the code like below:\n    ....\n    440: (b7) r1 \u003d 15\n    441: (05) goto pc+73\n    515: (79) r6 \u003d *(u64 *)(r10 -152)\n    516: (bf) r7 \u003d r10\n    517: (07) r7 +\u003d -112\n    518: (bf) r2 \u003d r7\n    519: (0f) r2 +\u003d r1\n    520: (71) r1 \u003d *(u8 *)(r8 +0)\n    521: (73) *(u8 *)(r2 +45) \u003d r1\n    ....\n    and the verifier complains \"R2 invalid mem access \u0027inv\u0027\" for insn #521.\n    This is because verifier marks register r2 as unknown value after #519\n    where r2 is a stack pointer and r1 holds a constant value.\n\n    Teach verifier to recognize \"stack_ptr + imm\" and\n    \"stack_ptr + reg with const val\" as valid stack_ptr with new offset.\n\n    Signed-off-by: Yonghong Song \u003cyhs@fb.com\u003e\n    Acked-by: Martin KaFai Lau \u003ckafai@fb.com\u003e\n    Acked-by: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Signed-off-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Greg Kroah-Hartman \u003cgregkh@linuxfoundation.org\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit bdb091c0307c84604a9b658cf4b67e05da12c1f1\nAuthor: Alexei Starovoitov \u003cast@fb.com\u003e\nDate:   Fri Mar 24 15:57:33 2017 -0700\n\n    bpf: improve verifier packet range checks\n\n    [ Upstream commit b1977682a3858b5584ffea7cfb7bd863f68db18d ]\n\n    llvm can optimize the \u0027if (ptr \u003e data_end)\u0027 checks to be in the order\n    slightly different than the original C code which will confuse verifier.\n    Like:\n    if (ptr + 16 \u003e data_end)\n      return TC_ACT_SHOT;\n    // may be followed by\n    if (ptr + 14 \u003e data_end)\n      return TC_ACT_SHOT;\n    while llvm can see that \u0027ptr\u0027 is valid for all 16 bytes,\n    the verifier could not.\n    Fix verifier logic to account for such case and add a test.\n\n    Reported-by: Huapeng Zhou \u003chzhou@fb.com\u003e\n    Fixes: 969bf05eb3ce (\"bpf: direct packet access\")\n    Signed-off-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Acked-by: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Acked-by: Martin KaFai Lau \u003ckafai@fb.com\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Greg Kroah-Hartman \u003cgregkh@linuxfoundation.org\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit dcb8b00abbd3935d4b54c7abd0adb54a3a6d1546\nAuthor: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\nDate:   Sun Dec 18 01:52:59 2016 +0100\n\n    bpf: fix mark_reg_unknown_value for spilled regs on map value marking\n\n    [ Upstream commit 6760bf2ddde8ad64f8205a651223a93de3a35494 ]\n\n    Martin reported a verifier issue that hit the BUG_ON() for his\n    test case in the mark_reg_unknown_value() function:\n\n      [  202.861380] kernel BUG at kernel/bpf/verifier.c:467!\n      [...]\n      [  203.291109] Call Trace:\n      [  203.296501]  [\u003cffffffff811364d5\u003e] mark_map_reg+0x45/0x50\n      [  203.308225]  [\u003cffffffff81136558\u003e] mark_map_regs+0x78/0x90\n      [  203.320140]  [\u003cffffffff8113938d\u003e] do_check+0x226d/0x2c90\n      [  203.331865]  [\u003cffffffff8113a6ab\u003e] bpf_check+0x48b/0x780\n      [  203.343403]  [\u003cffffffff81134c8e\u003e] bpf_prog_load+0x27e/0x440\n      [  203.355705]  [\u003cffffffff8118a38f\u003e] ? handle_mm_fault+0x11af/0x1230\n      [  203.369158]  [\u003cffffffff812d8188\u003e] ? security_capable+0x48/0x60\n      [  203.382035]  [\u003cffffffff811351a4\u003e] SyS_bpf+0x124/0x960\n      [  203.393185]  [\u003cffffffff810515f6\u003e] ? __do_page_fault+0x276/0x490\n      [  203.406258]  [\u003cffffffff816db320\u003e] entry_SYSCALL_64_fastpath+0x13/0x94\n\n    This issue got uncovered after the fix in a08dd0da5307 (\"bpf: fix\n    regression on verifier pruning wrt map lookups\"). The reason why it\n    wasn\u0027t noticed before was, because as mentioned in a08dd0da5307,\n    mark_map_regs() was doing the id matching incorrectly based on the\n    uncached regs[regno].id. So, in the first loop, we walked all regs\n    and as soon as we found regno \u003d\u003d i, then this reg\u0027s id was cleared\n    when calling mark_reg_unknown_value() thus that every subsequent\n    register was probed against id of 0 (which, in combination with the\n    PTR_TO_MAP_VALUE_OR_NULL type is an invalid condition that no other\n    register state can hold), and therefore wasn\u0027t type transitioned such\n    as in the spilled register case for the second loop.\n\n    Now since that got fixed, it turned out that 57a09bf0a416 (\"bpf:\n    Detect identical PTR_TO_MAP_VALUE_OR_NULL registers\") used\n    mark_reg_unknown_value() incorrectly for the spilled regs, and thus\n    hitting the BUG_ON() in some cases due to regno \u003e\u003d MAX_BPF_REG.\n\n    Although spilled regs have the same type as the non-spilled regs\n    for the verifier state, that is, struct bpf_reg_state, they are\n    semantically different from the non-spilled regs. In other words,\n    there can be up to 64 (MAX_BPF_STACK / BPF_REG_SIZE) spilled regs\n    in the stack, for example, register R\u003cx\u003e could have been spilled by\n    the program to stack location X, Y, Z, and in mark_map_regs() we\n    need to scan these stack slots of type STACK_SPILL for potential\n    registers that we have to transition from PTR_TO_MAP_VALUE_OR_NULL.\n    Therefore, depending on the location, the spilled_regs regno can\n    be a lot higher than just MAX_BPF_REG\u0027s value since we operate on\n    stack instead. The reset in mark_reg_unknown_value() itself is\n    just fine, only that the BUG_ON() was inappropriate for this. Fix\n    it by making a __mark_reg_unknown_value() version that can be\n    called from mark_map_reg() generically; we know for the non-spilled\n    case that the regno is always \u003c MAX_BPF_REG anyway.\n\n    Fixes: 57a09bf0a416 (\"bpf: Detect identical PTR_TO_MAP_VALUE_OR_NULL registers\")\n    Reported-by: Martin KaFai Lau \u003ckafai@fb.com\u003e\n    Signed-off-by: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Acked-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Greg Kroah-Hartman \u003cgregkh@linuxfoundation.org\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit da6b22793305d70094696ecf00bbfcb52eab4aa7\nAuthor: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\nDate:   Thu Dec 15 01:30:06 2016 +0100\n\n    bpf: fix regression on verifier pruning wrt map lookups\n\n    [ Upstream commit a08dd0da5307ba01295c8383923e51e7997c3576 ]\n\n    Commit 57a09bf0a416 (\"bpf: Detect identical PTR_TO_MAP_VALUE_OR_NULL\n    registers\") introduced a regression where existing programs stopped\n    loading due to reaching the verifier\u0027s maximum complexity limit,\n    whereas prior to this commit they were loading just fine; the affected\n    program has roughly 2k instructions.\n\n    What was found is that state pruning couldn\u0027t be performed effectively\n    anymore due to mismatches of the verifier\u0027s register state, in particular\n    in the id tracking. It doesn\u0027t mean that 57a09bf0a416 is incorrect per\n    se, but rather that verifier needs to perform a lot more work for the\n    same program with regards to involved map lookups.\n\n    Since commit 57a09bf0a416 is only about tracking registers with type\n    PTR_TO_MAP_VALUE_OR_NULL, the id is only needed to follow registers\n    until they are promoted through pattern matching with a NULL check to\n    either PTR_TO_MAP_VALUE or UNKNOWN_VALUE type. After that point, the\n    id becomes irrelevant for the transitioned types.\n\n    For UNKNOWN_VALUE, id is already reset to 0 via mark_reg_unknown_value(),\n    but not so for PTR_TO_MAP_VALUE where id is becoming stale. It\u0027s even\n    transferred further into other types that don\u0027t make use of it. Among\n    others, one example is where UNKNOWN_VALUE is set on function call\n    return with RET_INTEGER return type.\n\n    states_equal() will then fall through the memcmp() on register state;\n    note that the second memcmp() uses offsetofend(), so the id is part of\n    that since d2a4dd37f6b4 (\"bpf: fix state equivalence\"). But the bisect\n    pointed already to 57a09bf0a416, where we really reach beyond complexity\n    limit. What I found was that states_equal() often failed in this\n    case due to id mismatches in spilled regs with registers in type\n    PTR_TO_MAP_VALUE. Unlike non-spilled regs, spilled regs just perform\n    a memcmp() on their reg state and don\u0027t have any other optimizations\n    in place, therefore also id was relevant in this case for making a\n    pruning decision.\n\n    We can safely reset id to 0 as well when converting to PTR_TO_MAP_VALUE.\n    For the affected program, it resulted in a ~17 fold reduction of\n    complexity and let the program load fine again. Selftest suite also\n    runs fine. The only other place where env-\u003eid_gen is used currently is\n    through direct packet access, but for these cases id is long living, thus\n    a different scenario.\n\n    Also, the current logic in mark_map_regs() is not fully correct when\n    marking NULL branch with UNKNOWN_VALUE. We need to cache the destination\n    reg\u0027s id in any case. Otherwise, once we marked that reg as UNKNOWN_VALUE,\n    it\u0027s id is reset and any subsequent registers that hold the original id\n    and are of type PTR_TO_MAP_VALUE_OR_NULL won\u0027t be marked UNKNOWN_VALUE\n    anymore, since mark_map_reg() reuses the uncached regs[regno].id that\n    was just overridden. Note, we don\u0027t need to cache it outside of\n    mark_map_regs(), since it\u0027s called once on this_branch and the other\n    time on other_branch, which are both two independent verifier states.\n    A test case for this is added here, too.\n\n    Fixes: 57a09bf0a416 (\"bpf: Detect identical PTR_TO_MAP_VALUE_OR_NULL registers\")\n    Signed-off-by: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Acked-by: Thomas Graf \u003ctgraf@suug.ch\u003e\n    Acked-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Greg Kroah-Hartman \u003cgregkh@linuxfoundation.org\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 33fbd369cbe34c2136b5fb69d112eb3071769e65\nAuthor: Alexei Starovoitov \u003cast@fb.com\u003e\nDate:   Wed Dec 7 10:57:59 2016 -0800\n\n    bpf: fix state equivalence\n\n    [ Upstream commit d2a4dd37f6b41fbcad76efbf63124eb3126c66fe ]\n\n    Commmits 57a09bf0a416 (\"bpf: Detect identical PTR_TO_MAP_VALUE_OR_NULL registers\")\n    and 484611357c19 (\"bpf: allow access into map value arrays\") by themselves\n    are correct, but in combination they make state equivalence ignore \u0027id\u0027 field\n    of the register state which can lead to accepting invalid program.\n\n    Fixes: 57a09bf0a416 (\"bpf: Detect identical PTR_TO_MAP_VALUE_OR_NULL registers\")\n    Fixes: 484611357c19 (\"bpf: allow access into map value arrays\")\n    Signed-off-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Acked-by: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Acked-by: Thomas Graf \u003ctgraf@suug.ch\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Greg Kroah-Hartman \u003cgregkh@linuxfoundation.org\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 79f29ec3164b75194c9373aa384f2bec39bdca12\nAuthor: Thomas Graf \u003ctgraf@suug.ch\u003e\nDate:   Tue Oct 18 19:51:19 2016 +0200\n\n    bpf: Detect identical PTR_TO_MAP_VALUE_OR_NULL registers\n\n    [ Upstream commit 57a09bf0a416700676e77102c28f9cfcb48267e0 ]\n\n    A BPF program is required to check the return register of a\n    map_elem_lookup() call before accessing memory. The verifier keeps\n    track of this by converting the type of the result register from\n    PTR_TO_MAP_VALUE_OR_NULL to PTR_TO_MAP_VALUE after a conditional\n    jump ensures safety. This check is currently exclusively performed\n    for the result register 0.\n\n    In the event the compiler reorders instructions, BPF_MOV64_REG\n    instructions may be moved before the conditional jump which causes\n    them to keep their type PTR_TO_MAP_VALUE_OR_NULL to which the\n    verifier objects when the register is accessed:\n\n    0: (b7) r1 \u003d 10\n    1: (7b) *(u64 *)(r10 -8) \u003d r1\n    2: (bf) r2 \u003d r10\n    3: (07) r2 +\u003d -8\n    4: (18) r1 \u003d 0x59c00000\n    6: (85) call 1\n    7: (bf) r4 \u003d r0\n    8: (15) if r0 \u003d\u003d 0x0 goto pc+1\n     R0\u003dmap_value(ks\u003d8,vs\u003d8) R4\u003dmap_value_or_null(ks\u003d8,vs\u003d8) R10\u003dfp\n    9: (7a) *(u64 *)(r4 +0) \u003d 0\n    R4 invalid mem access \u0027map_value_or_null\u0027\n\n    This commit extends the verifier to keep track of all identical\n    PTR_TO_MAP_VALUE_OR_NULL registers after a map_elem_lookup() by\n    assigning them an ID and then marking them all when the conditional\n    jump is observed.\n\n    Signed-off-by: Thomas Graf \u003ctgraf@suug.ch\u003e\n    Reviewed-by: Josef Bacik \u003cjbacik@fb.com\u003e\n    Acked-by: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Acked-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Greg Kroah-Hartman \u003cgregkh@linuxfoundation.org\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 50630b1091b43615227714bb338cb6ad6cbbf514\nAuthor: Josef Bacik \u003cjbacik@fb.com\u003e\nDate:   Tue Nov 29 12:27:09 2016 -0500\n\n    bpf: fix states equal logic for varlen access\n\n    If we have a branch that looks something like this\n\n    int foo \u003d map-\u003evalue;\n    if (condition) {\n      foo +\u003d blah;\n    } else {\n      foo \u003d bar;\n    }\n    map-\u003earray[foo] \u003d baz;\n\n    We will incorrectly assume that the !condition branch is equal to the condition\n    branch as the register for foo will be UNKNOWN_VALUE in both cases.  We need to\n    adjust this logic to only do this if we didn\u0027t do a varlen access after we\n    processed the !condition branch, otherwise we have different ranges and need to\n    check the other branch as well.\n\n    Fixes: 484611357c19 (\"bpf: allow access into map value arrays\")\n    Reported-by: Jann Horn \u003cjannh@google.com\u003e\n    Signed-off-by: Josef Bacik \u003cjbacik@fb.com\u003e\n    Acked-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Acked-by: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 159be304d046a54e4829ad92d49bfc225bf426aa\nAuthor: Josef Bacik \u003cjbacik@fb.com\u003e\nDate:   Mon Nov 14 15:45:36 2016 -0500\n\n    bpf: fix range arithmetic for bpf map access\n\n    I made some invalid assumptions with BPF_AND and BPF_MOD that could result in\n    invalid accesses to bpf map entries.  Fix this up by doing a few things\n\n    1) Kill BPF_MOD support.  This doesn\u0027t actually get used by the compiler in real\n    life and just adds extra complexity.\n\n    2) Fix the logic for BPF_AND, don\u0027t allow AND of negative numbers and set the\n    minimum value to 0 for positive AND\u0027s.\n\n    3) Don\u0027t do operations on the ranges if they are set to the limits, as they are\n    by definition undefined, and allowing arithmetic operations on those values\n    could make them appear valid when they really aren\u0027t.\n\n    This fixes the testcase provided by Jann as well as a few other theoretical\n    problems.\n\n    Reported-by: Jann Horn \u003cjannh@google.com\u003e\n    Signed-off-by: Josef Bacik \u003cjbacik@fb.com\u003e\n    Acked-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 6b47191dd38f600cf718937443ae37d6094d5cbb\nAuthor: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\nDate:   Fri Nov 4 00:01:19 2016 +0100\n\n    bpf: fix htab map destruction when extra reserve is in use\n\n    Commit a6ed3ea65d98 (\"bpf: restore behavior of bpf_map_update_elem\")\n    added an extra per-cpu reserve to the hash table map to restore old\n    behaviour from pre prealloc times. When non-prealloc is in use for a\n    map, then problem is that once a hash table extra element has been\n    linked into the hash-table, and the hash table is destroyed due to\n    refcount dropping to zero, then htab_map_free() -\u003e delete_all_elements()\n    will walk the whole hash table and drop all elements via htab_elem_free().\n    The problem is that the element from the extra reserve is first fed\n    to the wrong backend allocator and eventually freed twice.\n\n    Fixes: a6ed3ea65d98 (\"bpf: restore behavior of bpf_map_update_elem\")\n    Reported-by: Dmitry Vyukov \u003cdvyukov@google.com\u003e\n    Signed-off-by: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Acked-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 5e1efbe7698fd2296a260776cee8678c401adaa1\nAuthor: Josef Bacik \u003cjbacik@fb.com\u003e\nDate:   Wed Sep 28 10:54:32 2016 -0400\n\n    bpf: allow access into map value arrays\n\n    Suppose you have a map array value that is something like this\n\n    struct foo {\n    \tunsigned iter;\n    \tint array[SOME_CONSTANT];\n    };\n\n    You can easily insert this into an array, but you cannot modify the contents of\n    foo-\u003earray[] after the fact.  This is because we have no way to verify we won\u0027t\n    go off the end of the array at verification time.  This patch provides a start\n    for this work.  We accomplish this by keeping track of a minimum and maximum\n    value a register could be while we\u0027re checking the code.  Then at the time we\n    try to do an access into a MAP_VALUE we verify that the maximum offset into that\n    region is a valid access into that memory region.  So in practice, code such as\n    this\n\n    unsigned index \u003d 0;\n\n    if (foo-\u003eiter \u003e\u003d SOME_CONSTANT)\n    \tfoo-\u003eiter \u003d index;\n    else\n    \tindex \u003d foo-\u003eiter++;\n    foo-\u003earray[index] \u003d bar;\n\n    would be allowed, as we can verify that index will always be between 0 and\n    SOME_CONSTANT-1.  If you wish to use signed values you\u0027ll have to have an extra\n    check to make sure the index isn\u0027t less than 0, or do something like index %\u003d\n    SOME_CONSTANT.\n\n    Signed-off-by: Josef Bacik \u003cjbacik@fb.com\u003e\n    Acked-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 45e550d9a96c8a67da70f4e7dace7b03e7b09e1e\nAuthor: Shaohua Li \u003cshli@fb.com\u003e\nDate:   Tue Sep 27 08:42:41 2016 -0700\n\n    bpf: clean up put_cpu_var usage\n\n    put_cpu_var takes the percpu data, not the data returned from\n    get_cpu_var.\n\n    This doesn\u0027t change the behavior.\n\n    Cc: Tejun Heo \u003ctj@kernel.org\u003e\n    Cc: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Signed-off-by: Shaohua Li \u003cshli@fb.com\u003e\n    Acked-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Acked-by: Tejun Heo \u003ctj@kernel.org\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 3590557296d6ee85588ae323b479e2a352499b2a\nAuthor: Mickaël Salaün \u003cmic@digikod.net\u003e\nDate:   Sat Sep 24 20:01:50 2016 +0200\n\n    bpf: Set register type according to is_valid_access()\n\n    This prevent future potential pointer leaks when an unprivileged eBPF\n    program will read a pointer value from its context. Even if\n    is_valid_access() returns a pointer type, the eBPF verifier replace it\n    with UNKNOWN_VALUE. The register value that contains a kernel address is\n    then allowed to leak. Moreover, this fix allows unprivileged eBPF\n    programs to use functions with (legitimate) pointer arguments.\n\n    Not an issue currently since reg_type is only set for PTR_TO_PACKET or\n    PTR_TO_PACKET_END in XDP and TC programs that can only be loaded as\n    privileged. For now, the only unprivileged eBPF program allowed is for\n    socket filtering and all the types from its context are UNKNOWN_VALUE.\n    However, this fix is important for future unprivileged eBPF programs\n    which could use pointers in their context.\n\n    Signed-off-by: Mickaël Salaün \u003cmic@digikod.net\u003e\n    Cc: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Cc: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Acked-by: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Acked-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 40b1393136917c9831e09614cf7be5952e6458e4\nAuthor: Jakub Kicinski \u003cjakub.kicinski@netronome.com\u003e\nDate:   Wed Sep 21 11:43:59 2016 +0100\n\n    bpf: recognize 64bit immediate loads as consts\n\n    When running as parser interpret BPF_LD | BPF_IMM | BPF_DW\n    instructions as loading CONST_IMM with the value stored\n    in imm.  The verifier will continue not recognizing those\n    due to concerns about search space/program complexity\n    increase.\n\n    Signed-off-by: Jakub Kicinski \u003cjakub.kicinski@netronome.com\u003e\n    Acked-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Acked-by: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 4e3cc7eaa89d3c68cd1254779b2c87a32c151cc1\nAuthor: Jakub Kicinski \u003cjakub.kicinski@netronome.com\u003e\nDate:   Wed Sep 21 11:43:58 2016 +0100\n\n    bpf: enable non-core use of the verfier\n\n    Advanced JIT compilers and translators may want to use\n    eBPF verifier as a base for parsers or to perform custom\n    checks and validations.\n\n    Add ability for external users to invoke the verifier\n    and provide callbacks to be invoked for every intruction\n    checked.  For now only add most basic callback for\n    per-instruction pre-interpretation checks is added.  More\n    advanced users may also like to have per-instruction post\n    callback and state comparison callback.\n\n    Signed-off-by: Jakub Kicinski \u003cjakub.kicinski@netronome.com\u003e\n    Acked-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 8ebc5af0ed1d2161e96c7b932df95771c3c7bd18\nAuthor: Jakub Kicinski \u003cjakub.kicinski@netronome.com\u003e\nDate:   Wed Sep 21 11:43:57 2016 +0100\n\n    bpf: expose internal verfier structures\n\n    Move verifier\u0027s internal structures to a header file and\n    prefix their names with bpf_ to avoid potential namespace\n    conflicts.  Those structures will soon be used by external\n    analyzers.\n\n    Signed-off-by: Jakub Kicinski \u003cjakub.kicinski@netronome.com\u003e\n    Acked-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Acked-by: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 3063e2e078c18f35fd3ca541c0d94d5ea01e08e5\nAuthor: Jakub Kicinski \u003cjakub.kicinski@netronome.com\u003e\nDate:   Wed Sep 21 11:43:56 2016 +0100\n\n    bpf: don\u0027t (ab)use instructions to store state\n\n    Storing state in reserved fields of instructions makes\n    it impossible to run verifier on programs already\n    marked as read-only. Allocate and use an array of\n    per-instruction state instead.\n\n    While touching the error path rename and move existing\n    jump target.\n\n    Suggested-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Signed-off-by: Jakub Kicinski \u003cjakub.kicinski@netronome.com\u003e\n    Acked-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Acked-by: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 73064ad9fddf5a1d2c2b3666b0b8fb71c53eff65\nAuthor: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\nDate:   Tue Sep 20 00:26:13 2016 +0200\n\n    bpf: direct packet write and access for helpers for clsact progs\n\n    This work implements direct packet access for helpers and direct packet\n    write in a similar fashion as already available for XDP types via commits\n    4acf6c0b84c9 (\"bpf: enable direct packet data write for xdp progs\") and\n    6841de8b0d03 (\"bpf: allow helpers access the packet directly\"), and as a\n    complementary feature to the already available direct packet read for tc\n    (cls/act) programs.\n\n    For enabling this, we need to introduce two helpers, bpf_skb_pull_data()\n    and bpf_csum_update(). The first is generally needed for both, read and\n    write, because they would otherwise only be limited to the current linear\n    skb head. Usually, when the data_end test fails, programs just bail out,\n    or, in the direct read case, use bpf_skb_load_bytes() as an alternative\n    to overcome this limitation. If such data sits in non-linear parts, we\n    can just pull them in once with the new helper, retest and eventually\n    access them.\n\n    At the same time, this also makes sure the skb is uncloned, which is, of\n    course, a necessary condition for direct write. As this needs to be an\n    invariant for the write part only, the verifier detects writes and adds\n    a prologue that is calling bpf_skb_pull_data() to effectively unclone the\n    skb from the very beginning in case it is indeed cloned. The heuristic\n    makes use of a similar trick that was done in 233577a22089 (\"net: filter:\n    constify detection of pkt_type_offset\"). This comes at zero cost for other\n    programs that do not use the direct write feature. Should a program use\n    this feature only sparsely and has read access for the most parts with,\n    for example, drop return codes, then such write action can be delegated\n    to a tail called program for mitigating this cost of potential uncloning\n    to a late point in time where it would have been paid similarly with the\n    bpf_skb_store_bytes() as well. Advantage of direct write is that the\n    writes are inlined whereas the helper cannot make any length assumptions\n    and thus needs to generate a call to memcpy() also for small sizes, as well\n    as cost of helper call itself with sanity checks are avoided. Plus, when\n    direct read is already used, we don\u0027t need to cache or perform rechecks\n    on the data boundaries (due to verifier invalidating previous checks for\n    helpers that change skb-\u003edata), so more complex programs using rewrites\n    can benefit from switching to direct read plus write.\n\n    For direct packet access to helpers, we save the otherwise needed copy into\n    a temp struct sitting on stack memory when use-case allows. Both facilities\n    are enabled via may_access_direct_pkt_data() in verifier. For now, we limit\n    this to map helpers and csum_diff, and can successively enable other helpers\n    where we find it makes sense. Helpers that definitely cannot be allowed for\n    this are those part of bpf_helper_changes_skb_data() since they can change\n    underlying data, and those that write into memory as this could happen for\n    packet typed args when still cloned. bpf_csum_update() helper accommodates\n    for the fact that we need to fixup checksum_complete when using direct write\n    instead of bpf_skb_store_bytes(), meaning the programs can use available\n    helpers like bpf_csum_diff(), and implement csum_add(), csum_sub(),\n    csum_block_add(), csum_block_sub() equivalents in eBPF together with the\n    new helper. A usage example will be provided for iproute2\u0027s examples/bpf/\n    directory.\n\n    Signed-off-by: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Acked-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 39d0ddc7063e15df6be785f1d5c23154c87a884f\nAuthor: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\nDate:   Tue Sep 20 00:26:12 2016 +0200\n\n    bpf, verifier: enforce larger zero range for pkt on overloading stack buffs\n\n    Current contract for the following two helper argument types is:\n\n      * ARG_CONST_STACK_SIZE: passed argument pair must be (ptr, \u003e0).\n      * ARG_CONST_STACK_SIZE_OR_ZERO: passed argument pair can be either\n        (NULL, 0) or (ptr, \u003e0).\n\n    With 6841de8b0d03 (\"bpf: allow helpers access the packet directly\"), we can\n    pass also raw packet data to helpers, so depending on the argument type\n    being PTR_TO_PACKET, we now either assert memory via check_packet_access()\n    or check_stack_boundary(). As a result, the tests in check_packet_access()\n    currently allow more than intended with regards to reg-\u003eimm.\n\n    Back in 969bf05eb3ce (\"bpf: direct packet access\"), check_packet_access()\n    was fine to ignore size argument since in check_mem_access() size was\n    bpf_size_to_bytes() derived and prior to the call to check_packet_access()\n    guaranteed to be larger than zero.\n\n    However, for the above two argument types, it currently means, we can have\n    a \u003c\u003d 0 size and thus breaking current guarantees for helpers. Enforce a\n    check for size \u003c\u003d 0 and bail out if so.\n\n    check_stack_boundary() doesn\u0027t have such an issue since it already tests\n    for access_size \u003c\u003d 0 and bails out, resp. access_size \u003d\u003d 0 in case of NULL\n    pointer passed when allowed.\n\n    Fixes: 6841de8b0d03 (\"bpf: allow helpers access the packet directly\")\n    Signed-off-by: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Acked-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit e1d63185c86a0ffa3b3facd49aca34c54152ce70\nAuthor: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\nDate:   Fri Sep 9 02:45:31 2016 +0200\n\n    bpf: add BPF_CALL_x macros for declaring helpers\n\n    This work adds BPF_CALL_\u003cn\u003e() macros and converts all the eBPF helper functions\n    to use them, in a similar fashion like we do with SYSCALL_DEFINE\u003cn\u003e() macros\n    that are used today. Motivation for this is to hide all the register handling\n    and all necessary casts from the user, so that it is done automatically in the\n    background when adding a BPF_CALL_\u003cn\u003e() call.\n\n    This makes current helpers easier to review, eases to write future helpers,\n    avoids getting the casting mess wrong, and allows for extending all helpers at\n    once (f.e. build time checks, etc). It also helps detecting more easily in\n    code reviews that unused registers are not instrumented in the code by accident,\n    breaking compatibility with existing programs.\n\n    BPF_CALL_\u003cn\u003e() internals are quite similar to SYSCALL_DEFINE\u003cn\u003e() ones with some\n    fundamental differences, for example, for generating the actual helper function\n    that carries all u64 regs, we need to fill unused regs, so that we always end up\n    with 5 u64 regs as an argument.\n\n    I reviewed several 0-5 generated BPF_CALL_\u003cn\u003e() variants of the .i results and\n    they look all as expected. No sparse issue spotted. We let this also sit for a\n    few days with Fengguang\u0027s kbuild test robot, and there were no issues seen. On\n    s390, it barked on the \"uses dynamic stack allocation\" notice, which is an old\n    one from bpf_perf_event_output{,_tp}() reappearing here due to the conversion\n    to the call wrapper, just telling that the perf raw record/frag sits on stack\n    (gcc with s390\u0027s -mwarn-dynamicstack), but that\u0027s all. Did various runtime tests\n    and they were fine as well. All eBPF helpers are now converted to use these\n    macros, getting rid of a good chunk of all the raw castings.\n\n    Signed-off-by: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Acked-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit a741b1a3a4aad31a0f7bcfc76d2bfb296a559422\nAuthor: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\nDate:   Fri Aug 5 00:11:13 2016 +0200\n\n    bpf: fix checksum for vlan push/pop helper\n\n    When having skbs on ingress with CHECKSUM_COMPLETE, tc BPF programs don\u0027t\n    push rcsum of mac header back in and after BPF run back pull out again as\n    opposed to some other subsystems (ovs, for example).\n\n    For cases like q-in-q, meaning when a vlan tag for offloading is already\n    present and we\u0027re about to push another one, then skb_vlan_push() pushes the\n    inner one into the skb, increasing mac header and skb_postpush_rcsum()\u0027ing\n    the 4 bytes vlan header diff. Likewise, for the reverse operation in\n    skb_vlan_pop() for the case where vlan header needs to be pulled out of the\n    skb, we\u0027re decreasing the mac header and skb_postpull_rcsum()\u0027ing the 4 bytes\n    rcsum of the vlan header that was removed.\n\n    However mangling the rcsum here will lead to hw csum failure for BPF case,\n    since we\u0027re pulling or pushing data that was not part of the current rcsum.\n    Changing tc BPF programs in general to push/pull rcsum around BPF_PROG_RUN()\n    is also not really an option since current behaviour is ABI by now, but apart\n    from that would also mean to do quite a bit of useless work in the sense that\n    usually 12 bytes need to be rcsum pushed/pulled also when we don\u0027t need to\n    touch this vlan related corner case. One way to fix it would be to push the\n    necessary rcsum fixup down into vlan helpers that are (mostly) slow-path\n    anyway.\n\n    Fixes: 4e10df9a60d9 (\"bpf: introduce bpf_skb_vlan_push/pop() helpers\")\n    Signed-off-by: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Acked-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 2de9eb76504bfb258f33bf9a2ca3766cf75b7ebd\nAuthor: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\nDate:   Fri Aug 5 00:11:12 2016 +0200\n\n    bpf: fix checksum fixups on bpf_skb_store_bytes\n\n    bpf_skb_store_bytes() invocations above L2 header need BPF_F_RECOMPUTE_CSUM\n    flag for updates, so that CHECKSUM_COMPLETE will be fixed up along the way.\n    Where we ran into an issue with bpf_skb_store_bytes() is when we did a\n    single-byte update on the IPv6 hoplimit despite using BPF_F_RECOMPUTE_CSUM\n    flag; simple ping via ICMPv6 triggered a hw csum failure as a result. The\n    underlying issue has been tracked down to a buffer alignment issue.\n\n    Meaning, that csum_partial() computations via skb_postpull_rcsum() and\n    skb_postpush_rcsum() pair invoked had a wrong result since they operated on\n    an odd address for the hoplimit, while other computations were done on an\n    even address. This mix doesn\u0027t work as-is with skb_postpull_rcsum(),\n    skb_postpush_rcsum() pair as it always expects at least half-word alignment\n    of input buffers, which is normally the case. Thus, instead of these helpers\n    using csum_sub() and (implicitly) csum_add(), we need to use csum_block_sub(),\n    csum_block_add(), respectively. For unaligned offsets, they rotate the sum\n    to align it to a half-word boundary again, otherwise they work the same as\n    csum_sub() and csum_add().\n\n    Adding __skb_postpull_rcsum(), __skb_postpush_rcsum() variants that take the\n    offset as an input and adapting bpf_skb_store_bytes() to them fixes the hw\n    csum failures again. The skb_postpull_rcsum(), skb_postpush_rcsum() helpers\n    use a 0 constant for offset so that the compiler optimizes the offset \u0026 1\n    test away and generates the same code as with csum_sub()/_add().\n\n    Fixes: 608cd71a9c7c (\"tc: bpf: generalize pedit action\")\n    Signed-off-by: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Acked-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 74f2ad88a2001b429108583f473fb09002a3203d\nAuthor: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\nDate:   Fri Aug 5 00:11:11 2016 +0200\n\n    bpf: also call skb_postpush_rcsum on xmit occasions\n\n    Follow-up to commit f8ffad69c9f8 (\"bpf: add skb_postpush_rcsum and fix\n    dev_forward_skb occasions\") to fix an issue for dev_queue_xmit() redirect\n    locations which need CHECKSUM_COMPLETE fixups on ingress.\n\n    For the same reasons as described in f8ffad69c9f8 already, we of course\n    also need this here, since dev_queue_xmit() on a veth device will let us\n    end up in the dev_forward_skb() helper again to cross namespaces.\n\n    Latter then calls into skb_postpull_rcsum() to pull out L2 header, so\n    that netif_rx_internal() sees CHECKSUM_COMPLETE as it is expected. That\n    is, CHECKSUM_COMPLETE on ingress covering L2 _payload_, not L2 headers.\n\n    Also here we have to address bpf_redirect() and bpf_clone_redirect().\n\n    Fixes: 3896d655f4d4 (\"bpf: introduce bpf_clone_redirect() helper\")\n    Fixes: 27b29f63058d (\"bpf: add bpf_redirect() helper\")\n    Signed-off-by: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Acked-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 593de6e6adeea502727171c2c241bfc3c5662600\nAuthor: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\nDate:   Fri Jun 10 21:19:06 2016 +0200\n\n    bpf: enforce recursion limit on redirects\n\n    Respect the stack\u0027s xmit_recursion limit for calls into dev_queue_xmit().\n    Currently, they are not handeled by the limiter when attached to clsact\u0027s\n    egress parent, for example, and a buggy program redirecting it to the\n    same device again could run into stack overflow eventually. It would be\n    good if we could notify an admin to give him a chance to react. We reuse\n    xmit_recursion instead of having one private to eBPF, so that the stack\u0027s\n    current recursion depth will be taken into account as well. Follow-up to\n    commit 3896d655f4d4 (\"bpf: introduce bpf_clone_redirect() helper\") and\n    27b29f63058d (\"bpf: add bpf_redirect() helper\").\n\n    Signed-off-by: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Acked-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 4d0e8d99feac1c3848a4e007a8a6a9303f7bd185\nAuthor: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\nDate:   Fri Sep 9 02:45:30 2016 +0200\n\n    bpf: add own ctx rewriter on ifindex for clsact progs\n\n    When fetching ifindex, we don\u0027t need to test dev for being NULL since\n    we\u0027re always guaranteed to have a valid dev for clsact programs. Thus,\n    avoid this test in fast path.\n\n    Signed-off-by: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Acked-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 74a3283dbe7ca4209a7e611d86d4c784a973ea55\nAuthor: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\nDate:   Thu Jul 14 18:08:04 2016 +0200\n\n    bpf, perf: split bpf_perf_event_output\n\n    Split the bpf_perf_event_output() helper as a preparation into\n    two parts. The new bpf_perf_event_output() will prepare the raw\n    record itself and test for unknown flags from BPF trace context,\n    where the __bpf_perf_event_output() does the core work. The\n    latter will be reused later on from bpf_event_output() directly.\n\n    Signed-off-by: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Acked-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit b58e4c2239acbcb2ff5dcc56bc2b809856ad1494\nAuthor: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\nDate:   Fri Sep 9 02:45:29 2016 +0200\n\n    bpf: add BPF_SIZEOF and BPF_FIELD_SIZEOF macros\n\n    Add BPF_SIZEOF() and BPF_FIELD_SIZEOF() macros to improve the code a bit\n    which otherwise often result in overly long bytes_to_bpf_size(sizeof())\n    and bytes_to_bpf_size(FIELD_SIZEOF()) lines. So place them into a macro\n    helper instead. Moreover, we currently have a BUILD_BUG_ON(BPF_FIELD_SIZEOF())\n    check in convert_bpf_extensions(), but we should rather make that generic\n    as well and add a BUILD_BUG_ON() test in all BPF_SIZEOF()/BPF_FIELD_SIZEOF()\n    users to detect any rewriter size issues at compile time. Note, there are\n    currently none, but we want to assert that it stays this way.\n\n    Signed-off-by: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Acked-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit a302ee2b7cb80604da18a75dc96e2938756f3165\nAuthor: Alexei Starovoitov \u003cast@fb.com\u003e\nDate:   Thu Sep 1 18:37:22 2016 -0700\n\n    bpf: introduce BPF_PROG_TYPE_PERF_EVENT program type\n\n    Introduce BPF_PROG_TYPE_PERF_EVENT programs that can be attached to\n    HW and SW perf events (PERF_TYPE_HARDWARE and PERF_TYPE_SOFTWARE\n    correspondingly in uapi/linux/perf_event.h)\n\n    The program visible context meta structure is\n    struct bpf_perf_event_data {\n        struct pt_regs regs;\n         __u64 sample_period;\n    };\n    which is accessible directly from the program:\n    int bpf_prog(struct bpf_perf_event_data *ctx)\n    {\n      ... ctx-\u003esample_period ...\n      ... ctx-\u003eregs.ip ...\n    }\n\n    The bpf verifier rewrites the accesses into kernel internal\n    struct bpf_perf_event_data_kern which allows changing\n    struct perf_sample_data without affecting bpf programs.\n    New fields can be added to the end of struct bpf_perf_event_data\n    in the future.\n\n    Signed-off-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Acked-by: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 31d26b1bb536fbbd788864e31c694af3b6b89e08\nAuthor: Alexei Starovoitov \u003cast@fb.com\u003e\nDate:   Thu Aug 11 18:17:18 2016 -0700\n\n    bpf: allow bpf_get_prandom_u32() to be used in tracing\n\n    bpf_get_prandom_u32() was initially introduced for socket filters\n    and later requested numberous times to be added to tracing bpf programs\n    for the same reason as in socket filters: to be able to randomly\n    select incoming events.\n\n    Signed-off-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Acked-by: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 106ac62208205f2031b76adef574ccc19521df1b\nAuthor: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\nDate:   Fri Sep 9 02:45:28 2016 +0200\n\n    bpf: minor cleanups in helpers\n\n    Some minor misc cleanups, f.e. use sizeof(__u32) instead of hardcoding\n    and in __bpf_skb_max_len(), I missed that we always have skb-\u003edev valid\n    anyway, so we can drop the unneeded test for dev; also few more other\n    misc bits addressed here.\n\n    Signed-off-by: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Acked-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit a3758f72995752ef07997b2443abc6da8378c4ed\nAuthor: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\nDate:   Thu Aug 18 01:00:41 2016 +0200\n\n    bpf: get rid of cgroup helper related ifdefs\n\n    As recently discussed during the task_under_cgroup_hierarchy() addition,\n    we should get rid of the ifdefs surrounding the bpf_skb_under_cgroup()\n    helper. If related functionality is not built-in, the helper cannot be\n    used anyway, which is also in line with what we do for all other helpers.\n\n    Signed-off-by: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Acked-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit c0d73a3b333e2408152b3d82e648b0e633d5dcc2\nAuthor: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\nDate:   Thu Aug 18 01:00:40 2016 +0200\n\n    bpf: enable event output helper also for xdp types\n\n    Follow-up to 555c8a8623a3 (\"bpf: avoid stack copy and use skb ctx for\n    event output\") for also adding the event output helper for XDP typed\n    programs. The event output helper has been very useful in particular for\n    debugging or event notification purposes, since it\u0027s much faster and\n    flexible than regular trace printk due to programmatically being able to\n    attach meta data. Same flags structure applies as with tc BPF programs.\n\n    Signed-off-by: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Acked-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 7e1deaa7436324342057ac24aad76790f3628962\nAuthor: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\nDate:   Thu Aug 18 01:00:39 2016 +0200\n\n    bpf: add bpf_skb_change_tail helper\n\n    This work adds a bpf_skb_change_tail() helper for tc BPF programs. The\n    basic idea is to expand or shrink the skb in a controlled manner. The\n    eBPF program can then rewrite the rest via helpers like bpf_skb_store_bytes(),\n    bpf_lX_csum_replace() and others rather than passing a raw buffer for\n    writing here.\n\n    bpf_skb_change_tail() is really a slow path helper and intended for\n    replies with f.e. ICMP control messages. Concept is similar to other\n    helpers like bpf_skb_change_proto() helper to keep the helper without\n    protocol specifics and let the BPF program mangle the remaining parts.\n    A flags field has been added and is reserved for now should we extend\n    the helper in future.\n\n    Signed-off-by: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Acked-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 90fdb69390d0040ab2e79c180abee4d198aa24e5\nAuthor: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\nDate:   Thu Aug 18 01:00:38 2016 +0200\n\n    bpf: use skb_pkt_type_ok helper in bpf_skb_change_type\n\n    Since we have a skb_pkt_type_ok() helper for checking the type before\n    mangling, make use of it instead of open coding. Follow-up to commit\n    8b10cab64c13 (\"net: simplify and make pkt_type_ok() available for other\n    users\") that came in after d2485c4242a8 (\"bpf: add bpf_skb_change_type\n    helper\").\n\n    Signed-off-by: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Acked-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit be03e60a4c0d045ae1f72daf1b31bc8876531ae5\nAuthor: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\nDate:   Thu Aug 11 21:38:37 2016 +0200\n\n    bpf: fix write helpers with regards to non-linear parts\n\n    Fix the bpf_try_make_writable() helper and all call sites we have in BPF,\n    it\u0027s currently defect with regards to skbs when the write_len spans into\n    non-linear parts, no matter if cloned or not.\n\n    There are multiple issues at once. First, using skb_store_bits() is not\n    correct since even if we have a cloned skb, page frags can still be shared.\n    To really make them private, we need to pull them in via __pskb_pull_tail()\n    first, which also gets us a private head via pskb_expand_head() implicitly.\n\n    This is for helpers like bpf_skb_store_bytes(), bpf_l3_csum_replace(),\n    bpf_l4_csum_replace(). Really, the only thing reasonable and working here\n    is to call skb_ensure_writable() before any write operation. Meaning, via\n    pskb_may_pull() it makes sure that parts we want to access are pulled in and\n    if not does so plus unclones the skb implicitly. If our write_len still fits\n    the headlen and we\u0027re cloned and our header of the clone is not writable,\n    then we need to make a private copy via pskb_expand_head(). skb_store_bits()\n    is a bit misleading and only safe to store into non-linear data in different\n    contexts such as 357b40a18b04 (\"[IPV6]: IPV6_CHECKSUM socket option can\n    corrupt kernel memory\").\n\n    For above BPF helper functions, it means after fixed bpf_try_make_writable(),\n    we\u0027ve pulled in enough, so that we operate always based on skb-\u003edata. Thus,\n    the call to skb_header_pointer() and skb_store_bits() becomes superfluous.\n    In bpf_skb_store_bytes(), the len check is unnecessary too since it can\n    only pass in maximum of BPF stack size, so adding offset is guaranteed to\n    never overflow. Also bpf_l3/4_csum_replace() helpers must test for proper\n    offset alignment since they use __sum16 pointer for writing resulting csum.\n\n    The remaining helpers that change skb data not discussed here yet are\n    bpf_skb_vlan_push(), bpf_skb_vlan_pop() and bpf_skb_change_proto(). The\n    vlan helpers internally call either skb_ensure_writable() (pop case) and\n    skb_cow_head() (push case, for head expansion), respectively. Similarly,\n    bpf_skb_proto_xlat() takes care to not mangle page frags.\n\n    Fixes: 608cd71a9c7c (\"tc: bpf: generalize pedit action\")\n    Fixes: 91bc4822c3d6 (\"tc: bpf: add checksum helpers\")\n    Fixes: 3697649ff29e (\"bpf: try harder on clones when writing into skb\")\n    Signed-off-by: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Acked-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit d7815bc85a8a8de1c8514eff20bd1c7b983b4074\nAuthor: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\nDate:   Tue Sep 20 00:26:14 2016 +0200\n\n    bpf: add test cases for direct packet access\n\n    Add couple of test cases for direct write and the negative size issue, and\n    also adjust the direct packet access test4 since it asserts that writes are\n    not possible, but since we\u0027ve just added support for writes, we need to\n    invert the verdict to ACCEPT, of course. Summary: 133 PASSED, 0 FAILED.\n\n    Signed-off-by: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Acked-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 93977d093f0affaf99a7db2099f7cbbebf3a5d2e\nAuthor: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\nDate:   Thu Sep 8 01:03:42 2016 +0200\n\n    bpf: fix range propagation on direct packet access\n\n    LLVM can generate code that tests for direct packet access via\n    skb-\u003edata/data_end in a way that currently gets rejected by the\n    verifier, example:\n\n      [...]\n       7: (61) r3 \u003d *(u32 *)(r6 +80)\n       8: (61) r9 \u003d *(u32 *)(r6 +76)\n       9: (bf) r2 \u003d r9\n      10: (07) r2 +\u003d 54\n      11: (3d) if r3 \u003e\u003d r2 goto pc+12\n       R1\u003dinv R2\u003dpkt(id\u003d0,off\u003d54,r\u003d0) R3\u003dpkt_end R4\u003dinv R6\u003dctx\n       R9\u003dpkt(id\u003d0,off\u003d0,r\u003d0) R10\u003dfp\n      12: (18) r4 \u003d 0xffffff7a\n      14: (05) goto pc+430\n      [...]\n\n      from 11 to 24: R1\u003dinv R2\u003dpkt(id\u003d0,off\u003d54,r\u003d0) R3\u003dpkt_end R4\u003dinv\n                     R6\u003dctx R9\u003dpkt(id\u003d0,off\u003d0,r\u003d0) R10\u003dfp\n      24: (7b) *(u64 *)(r10 -40) \u003d r1\n      25: (b7) r1 \u003d 0\n      26: (63) *(u32 *)(r6 +56) \u003d r1\n      27: (b7) r2 \u003d 40\n      28: (71) r8 \u003d *(u8 *)(r9 +20)\n      invalid access to packet, off\u003d20 size\u003d1, R9(id\u003d0,off\u003d0,r\u003d0)\n\n    The reason why this gets rejected despite a proper test is that we\n    currently call find_good_pkt_pointers() only in case where we detect\n    tests like rX \u003e pkt_end, where rX is of type pkt(id\u003dY,off\u003dZ,r\u003d0) and\n    derived, for example, from a register of type pkt(id\u003dY,off\u003d0,r\u003d0)\n    pointing to skb-\u003edata. find_good_pkt_pointers() then fills the range\n    in the current branch to pkt(id\u003dY,off\u003d0,r\u003dZ) on success.\n\n    For above case, we need to extend that to recognize pkt_end \u003e\u003d rX\n    pattern and mark the other branch that is taken on success with the\n    appropriate pkt(id\u003dY,off\u003d0,r\u003dZ) type via find_good_pkt_pointers().\n    Since eBPF operates on BPF_JGT (\u003e) and BPF_JGE (\u003e\u003d), these are the\n    only two practical options to test for from what LLVM could have\n    generated, since there\u0027s no such thing as BPF_JLT (\u003c) or BPF_JLE (\u003c\u003d)\n    that we would need to take into account as well.\n\n    After the fix:\n\n      [...]\n       7: (61) r3 \u003d *(u32 *)(r6 +80)\n       8: (61) r9 \u003d *(u32 *)(r6 +76)\n       9: (bf) r2 \u003d r9\n      10: (07) r2 +\u003d 54\n      11: (3d) if r3 \u003e\u003d r2 goto pc+12\n       R1\u003dinv R2\u003dpkt(id\u003d0,off\u003d54,r\u003d0) R3\u003dpkt_end R4\u003dinv R6\u003dctx\n       R9\u003dpkt(id\u003d0,off\u003d0,r\u003d0) R10\u003dfp\n      12: (18) r4 \u003d 0xffffff7a\n      14: (05) goto pc+430\n      [...]\n\n      from 11 to 24: R1\u003dinv R2\u003dpkt(id\u003d0,off\u003d54,r\u003d54) R3\u003dpkt_end R4\u003dinv\n                     R6\u003dctx R9\u003dpkt(id\u003d0,off\u003d0,r\u003d54) R10\u003dfp\n      24: (7b) *(u64 *)(r10 -40) \u003d r1\n      25: (b7) r1 \u003d 0\n      26: (63) *(u32 *)(r6 +56) \u003d r1\n      27: (b7) r2 \u003d 40\n      28: (71) r8 \u003d *(u8 *)(r9 +20)\n      29: (bf) r1 \u003d r8\n      30: (25) if r8 \u003e 0x3c goto pc+47\n       R1\u003dinv56 R2\u003dimm40 R3\u003dpkt_end R4\u003dinv R6\u003dctx R8\u003dinv56\n       R9\u003dpkt(id\u003d0,off\u003d0,r\u003d54) R10\u003dfp\n      31: (b7) r1 \u003d 1\n      [...]\n\n    Verifier test cases are also added in this work, one that demonstrates\n    the mentioned example here and one that tries a bad packet access for\n    the current/fall-through branch (the one with types pkt(id\u003dX,off\u003dY,r\u003d0),\n    pkt(id\u003dX,off\u003d0,r\u003d0)), then a case with good and bad accesses, and two\n    with both test variants (\u003e, \u003e\u003d).\n\n    Fixes: 969bf05eb3ce (\"bpf: direct packet access\")\n    Signed-off-by: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Acked-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 99b8c16830283a3e33824c81f30e34e301bfabee\nAuthor: Aaron Yue \u003chaoxuany@fb.com\u003e\nDate:   Thu Aug 11 18:17:17 2016 -0700\n\n    samples/bpf: add verifier tests for the helper access to the packet\n\n    test various corner cases of the helper function access to the packet\n    via crafted XDP programs.\n\n    Signed-off-by: Aaron Yue \u003chaoxuany@fb.com\u003e\n    Signed-off-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Acked-by: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 5f9d8353899511f7c3768ba7cea16d5020f7168a\nAuthor: Alexei Starovoitov \u003cast@fb.com\u003e\nDate:   Thu May 5 19:49:15 2016 -0700\n\n    samples/bpf: add verifier tests\n\n    add few tests for \"pointer to packet\" logic of the verifier\n\n    Signed-off-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit fde3cb1df9819c40e2f9326c42d4db583b57a851\nAuthor: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\nDate:   Wed Apr 13 00:10:54 2016 +0200\n\n    bpf, samples: add test cases for raw stack\n\n    This adds test cases mostly around ARG_PTR_TO_RAW_STACK to check the\n    verifier behaviour.\n\n      [...]\n      #84 raw_stack: no skb_load_bytes OK\n      #85 raw_stack: skb_load_bytes, no init OK\n      #86 raw_stack: skb_load_bytes, init OK\n      #87 raw_stack: skb_load_bytes, spilled regs around bounds OK\n      #88 raw_stack: skb_load_bytes, spilled regs corruption OK\n      #89 raw_stack: skb_load_bytes, spilled regs corruption 2 OK\n      #90 raw_stack: skb_load_bytes, spilled regs + data OK\n      #91 raw_stack: skb_load_bytes, invalid access 1 OK\n      #92 raw_stack: skb_load_bytes, invalid access 2 OK\n      #93 raw_stack: skb_load_bytes, invalid access 3 OK\n      #94 raw_stack: skb_load_bytes, invalid access 4 OK\n      #95 raw_stack: skb_load_bytes, invalid access 5 OK\n      #96 raw_stack: skb_load_bytes, invalid access 6 OK\n      #97 raw_stack: skb_load_bytes, large access OK\n      Summary: 98 PASSED, 0 FAILED\n\n    Signed-off-by: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Acked-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit fc12f664c8e45b1c8d7ee364f89aa0a5088ae44b\nAuthor: Alexei Starovoitov \u003cast@fb.com\u003e\nDate:   Mon Mar 7 21:57:20 2016 -0800\n\n    samples/bpf: add map_flags to bpf loader\n\n    note old loader is compatible with new kernel.\n    map_flags are optional\n\n    Signed-off-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 0f9e9e27bb02b3d25b01c895734e1661d6305d25\nAuthor: tom.leiming@gmail.com \u003ctom.leiming@gmail.com\u003e\nDate:   Mon Feb 1 22:39:57 2016 -0800\n\n    samples/bpf: unit test for BPF_MAP_TYPE_PERCPU_ARRAY\n\n    A sanity test for BPF_MAP_TYPE_PERCPU_ARRAY\n\n    Signed-off-by: Ming Lei \u003ctom.leiming@gmail.com\u003e\n    Signed-off-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit c14ecd6ee61efc2b0f483dea6ff9239478437336\nAuthor: Alexei Starovoitov \u003cast@fb.com\u003e\nDate:   Mon Mar 7 21:57:18 2016 -0800\n\n    samples/bpf: make map creation more verbose\n\n    map creation is typically the first one to fail when rlimits are\n    too low, not enough memory, etc\n    Make this failure scenario more verbose\n\n    Signed-off-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 783c25cc972107cc38a257ae97d12916ee4473e2\nAuthor: Martin KaFai Lau \u003ckafai@fb.com\u003e\nDate:   Mon Feb 1 22:39:56 2016 -0800\n\n    samples/bpf: unit test for BPF_MAP_TYPE_PERCPU_HASH\n\n    A sanity test for BPF_MAP_TYPE_PERCPU_HASH.\n\n    Signed-off-by: Martin KaFai Lau \u003ckafai@fb.com\u003e\n    Signed-off-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 29d0150f7b3425b418f83de755aefc520df42c75\nAuthor: Alexei Starovoitov \u003cast@fb.com\u003e\nDate:   Thu Sep 1 18:37:23 2016 -0700\n\n    bpf: perf_event progs should only use preallocated maps\n\n    Make sure that BPF_PROG_TYPE_PERF_EVENT programs only use\n    preallocated hash maps, since doing memory allocation\n    in overflow_handler can crash depending on where nmi got triggered.\n\n    Signed-off-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Acked-by: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 95e4b804a67a53700c72e3357c1a07f0b0b624d7\nAuthor: Alexei Starovoitov \u003cast@fb.com\u003e\nDate:   Thu Sep 1 18:37:21 2016 -0700\n\n    bpf: support 8-byte metafield access\n\n    The verifier supported only 4-byte metafields in\n    struct __sk_buff and struct xdp_md. The metafields in upcoming\n    struct bpf_perf_event are 8-byte to match register width in struct pt_regs.\n    Teach verifier to recognize 8-byte metafield access.\n    The patch doesn\u0027t affect safety of sockets and xdp programs.\n    They check for 4-byte only ctx access before these conditions are hit.\n\n    Signed-off-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Acked-by: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 6307547affc3594aff1b2a23565f84b8e0fff69d\nAuthor: Alexei Starovoitov \u003cast@fb.com\u003e\nDate:   Thu Aug 11 18:17:16 2016 -0700\n\n    bpf: allow helpers access the packet directly\n\n    The helper functions like bpf_map_lookup_elem(map, key) were only\n    allowing \u0027key\u0027 to point to the initialized stack area.\n    That is causing performance degradation when programs need to process\n    millions of packets per second and need to copy contents of the packet\n    into the stack just to pass the stack pointer into the lookup() function.\n    Allow such helpers read from the packet directly.\n    All helpers that expect ARG_PTR_TO_MAP_KEY, ARG_PTR_TO_MAP_VALUE,\n    ARG_PTR_TO_STACK assume byte aligned pointer, so no alignment concerns,\n    only need to check that helper will not be accessing beyond\n    the packet range verified by the prior \u0027if (ptr \u003c data_end)\u0027 condition.\n    For now allow this feature for XDP programs only. Later it can be\n    relaxed for the clsact programs as well.\n\n    Signed-off-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Acked-by: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit ee8af304930b680ec746548bceb6baa5ce275027\nAuthor: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\nDate:   Fri Aug 12 22:17:17 2016 +0200\n\n    bpf: fix bpf_skb_in_cgroup helper naming\n\n    While hashing out BPF\u0027s current_task_under_cgroup helper bits, it came\n    to discussion that the skb_in_cgroup helper name was suboptimally chosen.\n\n    Tejun says:\n\n      So, I think in_cgroup should mean that the object is in that\n      particular cgroup while under_cgroup in the subhierarchy of that\n      cgroup. Let\u0027s rename the other subhierarchy test to under too. I\n      think that\u0027d be a lot less confusing going forward.\n\n      [...]\n\n      It\u0027s more intuitive and gives us the room to implement the real\n      \"in\" test if ever necessary in the future.\n\n    Since this touches uapi bits, we need to change this as long as v4.8\n    is not yet officially released. Thus, change the helper enum and rename\n    related bits.\n\n    Fixes: 4a482f34afcc (\"cgroup: bpf: Add bpf_skb_in_cgroup_proto\")\n    Reference: http://patchwork.ozlabs.org/patch/658500/\n    Suggested-by: Sargun Dhillon \u003csargun@sargun.me\u003e\n    Suggested-by: Tejun Heo \u003ctj@kernel.org\u003e\n    Signed-off-by: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Acked-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 19f4b8954ec2a51c65749410314db263ac7344f2\nAuthor: Martin KaFai Lau \u003ckafai@fb.com\u003e\nDate:   Thu Jun 30 10:28:45 2016 -0700\n\n    cgroup: bpf: Add an example to do cgroup checking in BPF\n\n    test_cgrp2_array_pin.c:\n    A userland program that creates a bpf_map (BPF_MAP_TYPE_GROUP_ARRAY),\n    pouplates/updates it with a cgroup2\u0027s backed fd and pins it to a\n    bpf-fs\u0027s file.  The pinned file can be loaded by tc and then used\n    by the bpf prog later.  This program can also update an existing pinned\n    array and it could be useful for debugging/testing purpose.\n\n    test_cgrp2_tc_kern.c:\n    A bpf prog which should be loaded by tc.  It is to demonstrate\n    the usage of bpf_skb_in_cgroup.\n\n    test_cgrp2_tc.sh:\n    A script that glues the test_cgrp2_array_pin.c and\n    test_cgrp2_tc_kern.c together.  The idea is like:\n    1. Load the test_cgrp2_tc_kern.o by tc\n    2. Use test_cgrp2_array_pin.c to populate a BPF_MAP_TYPE_CGROUP_ARRAY\n       with a cgroup fd\n    3. Do a \u0027ping -6 ff02::1%ve\u0027 to ensure the packet has been\n       dropped because of a match on the cgroup\n\n    Most of the lines in test_cgrp2_tc.sh is the boilerplate\n    to setup the cgroup/bpf-fs/net-devices/netns...etc.  It is\n    not bulletproof on errors but should work well enough and\n    give enough debug info if things did not go well.\n\n    Signed-off-by: Martin KaFai Lau \u003ckafai@fb.com\u003e\n    Cc: Alexei Starovoitov \u003cast@fb.com\u003e\n    Cc: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Cc: Tejun Heo \u003ctj@kernel.org\u003e\n    Acked-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 270d2dd08c88645792077255d89db2debff0ce49\nAuthor: Alexei Starovoitov \u003cast@fb.com\u003e\nDate:   Thu May 5 19:49:14 2016 -0700\n\n    samples/bpf: add \u0027pointer to packet\u0027 tests\n\n    parse_simple.c - packet parser exapmle with single length check that\n    filters out udp packets for port 9\n\n    parse_varlen.c - variable length parser that understand multiple vlan headers,\n    ipip, ipip6 and ip options to filter out udp or tcp packets on port 9.\n    The packet is parsed layer by layer with multitple length checks.\n\n    parse_ldabs.c - classic style of packet parsing using LD_ABS instruction.\n    Same functionality as parse_simple.\n\n    simple \u003d 24.1Mpps per core\n    varlen \u003d 22.7Mpps\n    ldabs  \u003d 21.4Mpps\n\n    Parser with LD_ABS instructions is slower than full direct access parser\n    which does more packet accesses and checks.\n\n    These examples demonstrate the choice bpf program authors can make between\n    flexibility of the parser vs speed.\n\n    Signed-off-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit c21e337c36d3167420e1e2a54a6266bce477a986\nAuthor: Alexei Starovoitov \u003cast@fb.com\u003e\nDate:   Wed Apr 6 18:43:31 2016 -0700\n\n    samples/bpf: add tracepoint vs kprobe performance tests\n\n    the first microbenchmark does\n    fd\u003dopen(\"/proc/self/comm\");\n    for() {\n      write(fd, \"test\");\n    }\n    and on 4 cpus in parallel:\n                                          writes per sec\n    base (no tracepoints, no kprobes)         930k\n    with kprobe at __set_task_comm()          420k\n    with tracepoint at task:task_rename       730k\n\n    For kprobe + full bpf program manully fetches oldcomm, newcomm via bpf_probe_read.\n    For tracepint bpf program does nothing, since arguments are copied by tracepoint.\n\n    2nd microbenchmark does:\n    fd\u003dopen(\"/dev/urandom\");\n    for() {\n      read(fd, buf);\n    }\n    and on 4 cpus in parallel:\n                                           reads per sec\n    base (no tracepoints, no kprobes)         300k\n    with kprobe at urandom_read()             279k\n    with tracepoint at random:urandom_read    290k\n\n    bpf progs attached to kprobe and tracepoint are noop.\n\n    Signed-off-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 9ec7ab3f84913d2c786ee8f04b6aa44bf3c6ec6e\nAuthor: Alexei Starovoitov \u003cast@fb.com\u003e\nDate:   Tue Mar 8 15:07:54 2016 -0800\n\n    samples/bpf: add map performance test\n\n    performance tests for hash map and per-cpu hash map\n    with and without pre-allocation\n\n    Signed-off-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 480b50b6342ef0c2a6ec9b0af06856503b27b4f8\nAuthor: Alexei Starovoitov \u003cast@fb.com\u003e\nDate:   Tue Mar 8 15:07:52 2016 -0800\n\n    samples/bpf: add bpf map stress test\n\n    this test calls bpf programs from different contexts:\n    from inside of slub, from rcu, from pretty much everywhere,\n    since it kprobes all spin_lock functions.\n    It stresses the bpf hash and percpu map pre-allocation,\n    deallocation logic and call_rcu mechanisms.\n    User space part adding more stress by walking and deleting map elements.\n\n    Note that due to nature bpf_load.c the earlier kprobe+bpf programs are\n    already active while loader loads new programs, creates new kprobes and\n    attaches them.\n\n    Signed-off-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit ed79a82bc4ab062c882dcc4ca9b1311ae45e7e91\nAuthor: Sargun Dhillon \u003csargun@sargun.me\u003e\nDate:   Fri Aug 12 08:56:52 2016 -0700\n\n    bpf: Add bpf_current_task_under_cgroup helper\n\n    This adds a bpf helper that\u0027s similar to the skb_in_cgroup helper to check\n    whether the probe is currently executing in the context of a specific\n    subset of the cgroupsv2 hierarchy. It does this based on membership test\n    for a cgroup arraymap. It is invalid to call this in an interrupt, and\n    it\u0027ll return an error. The helper is primarily to be used in debugging\n    activities for containers, where you may have multiple programs running in\n    a given top-level \"container\".\n\n    Signed-off-by: Sargun Dhillon \u003csargun@sargun.me\u003e\n    Cc: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Cc: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Cc: Tejun Heo \u003ctj@kernel.org\u003e\n    Acked-by: Tejun Heo \u003ctj@kernel.org\u003e\n    Acked-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Acked-by: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 672153c037d8cda96a8a7958916c1580e84684b7\nAuthor: Sargun Dhillon \u003csargun@sargun.me\u003e\nDate:   Mon Jul 25 05:54:46 2016 -0700\n\n    bpf: Add bpf_probe_write_user BPF helper to be called in tracers\n\n    This allows user memory to be written to during the course of a kprobe.\n    It shouldn\u0027t be used to implement any kind of security mechanism\n    because of TOC-TOU attacks, but rather to debug, divert, and\n    manipulate execution of semi-cooperative processes.\n\n    Although it uses probe_kernel_write, we limit the address space\n    the probe can write into by checking the space with access_ok.\n    We do this as opposed to calling copy_to_user directly, in order\n    to avoid sleeping. In addition we ensure the threads\u0027s current fs\n    / segment is USER_DS and the thread isn\u0027t exiting nor a kernel thread.\n\n    Given this feature is meant for experiments, and it has a risk of\n    crashing the system, and running programs, we print a warning on\n    when a proglet that attempts to use this helper is installed,\n    along with the pid and process name.\n\n    Signed-off-by: Sargun Dhillon \u003csargun@sargun.me\u003e\n    Cc: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Cc: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Acked-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit f08fa6f8429d8ed86c1a3832797205734912953e\nAuthor: Alexei Starovoitov \u003cast@fb.com\u003e\nDate:   Wed Feb 17 19:58:59 2016 -0800\n\n    samples/bpf: offwaketime example\n\n    This is simplified version of Brendan Gregg\u0027s offwaketime:\n    This program shows kernel stack traces and task names that were blocked and\n    \"off-CPU\", along with the stack traces and task names for the threads that woke\n    them, and the total elapsed time from when they blocked to when they were woken\n    up. The combined stacks, task names, and total time is summarized in kernel\n    context for efficiency.\n\n    Example:\n    $ sudo ./offwaketime | flamegraph.pl \u003e demo.svg\n    Open demo.svg in the browser as FlameGraph visualization.\n\n    Signed-off-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 3c6591feabb0fda49a8177224aaf5ddd97d8bbb7\nAuthor: Andrew Morton \u003cakpm@linux-foundation.org\u003e\nDate:   Mon Jul 18 15:50:58 2016 -0700\n\n    kernel/trace/bpf_trace.c: work around gcc-4.4.4 anon union initialization bug\n\n    kernel/trace/bpf_trace.c: In function \u0027bpf_event_output\u0027:\n    kernel/trace/bpf_trace.c:312: error: unknown field \u0027next\u0027 specified in initializer\n    kernel/trace/bpf_trace.c:312: warning: missing braces around initializer\n    kernel/trace/bpf_trace.c:312: warning: (near initialization for \u0027raw.frag.\u003canonymous\u003e\u0027)\n\n    Fixes: 555c8a8623a3a87 (\"bpf: avoid stack copy and use skb ctx for event output\")\n    Acked-by: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Cc: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Cc: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Andrew Morton \u003cakpm@linux-foundation.org\u003e\n    Acked-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit ffe455b3b57bc9f711a1d39dea9a71ddcf312a6d\nAuthor: Alexei Starovoitov \u003cast@fb.com\u003e\nDate:   Wed Jul 6 22:38:36 2016 -0700\n\n    bpf: introduce bpf_get_current_task() helper\n\n    over time there were multiple requests to access different data\n    structures and fields of task_struct current, so finally add\n    the helper to access \u0027current\u0027 as-is. Tracing bpf programs will do\n    the rest of walking the pointers via bpf_probe_read().\n    Note that current can be null and bpf program has to deal it with,\n    but even dumb passing null into bpf_probe_read() is still safe.\n\n    Suggested-by: Brendan Gregg \u003cbrendan.d.gregg@gmail.com\u003e\n    Signed-off-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Acked-by: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit c119ae5d7bbe41c58f88f365b258038922ecf239\nAuthor: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\nDate:   Sun Jul 3 01:28:47 2016 +0200\n\n    bpf: add bpf_get_hash_recalc helper\n\n    If skb_clear_hash() was invoked due to mangling of relevant headers and\n    BPF program needs skb-\u003ehash later on, we can add a helper to trigger hash\n    recalculation via bpf_get_hash_recalc().\n\n    The helper will return the newly retrieved hash directly, but later access\n    can also be done via skb context again through skb-\u003ehash directly (inline)\n    without needing to call the helper once more.\n\n    Signed-off-by: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Acked-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit de504ce5eeb34423bb46b5dab73c6349d4c1cead\nAuthor: Alexei Starovoitov \u003cast@fb.com\u003e\nDate:   Fri Aug 5 14:01:27 2016 -0700\n\n    bpf: restore behavior of bpf_map_update_elem\n\n    The introduction of pre-allocated hash elements inadvertently broke\n    the behavior of bpf hash maps where users expected to call\n    bpf_map_update_elem() without considering that the map can be full.\n    Some programs do:\n    old_value \u003d bpf_map_lookup_elem(map, key);\n    if (old_value) {\n      ... prepare new_value on stack ...\n      bpf_map_update_elem(map, key, new_value);\n    }\n    Before pre-alloc the update() for existing element would work even\n    in \u0027map full\u0027 condition. Restore this behavior.\n\n    The above program could have updated old_value in place instead of\n    update() which would be faster and most programs use that approach,\n    but sometimes the values are large and the programs use update()\n    helper to do atomic replacement of the element.\n    Note we cannot simply update element\u0027s value in-place like percpu\n    hash map does and have to allocate extra num_possible_cpu elements\n    and use this extra reserve when the map is full.\n\n    Fixes: 6c9059817432 (\"bpf: pre-allocate hash map elements\")\n    Signed-off-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 4ad633a8344518b7be441f047cf351591491f452\nAuthor: Jakub Kicinski \u003cjakub.kicinski@netronome.com\u003e\nDate:   Tue Aug 2 16:12:14 2016 +0100\n\n    bpf: fix method of PTR_TO_PACKET reg id generation\n\n    Using per-register incrementing ID can lead to\n    find_good_pkt_pointers() confusing registers which\n    have completely different values.  Consider example:\n\n    0: (bf) r6 \u003d r1\n    1: (61) r8 \u003d *(u32 *)(r6 +76)\n    2: (61) r0 \u003d *(u32 *)(r6 +80)\n    3: (bf) r7 \u003d r8\n    4: (07) r8 +\u003d 32\n    5: (2d) if r8 \u003e r0 goto pc+9\n     R0\u003dpkt_end R1\u003dctx R6\u003dctx R7\u003dpkt(id\u003d0,off\u003d0,r\u003d32) R8\u003dpkt(id\u003d0,off\u003d32,r\u003d32) R10\u003dfp\n    6: (bf) r8 \u003d r7\n    7: (bf) r9 \u003d r7\n    8: (71) r1 \u003d *(u8 *)(r7 +0)\n    9: (0f) r8 +\u003d r1\n    10: (71) r1 \u003d *(u8 *)(r7 +1)\n    11: (0f) r9 +\u003d r1\n    12: (07) r8 +\u003d 32\n    13: (2d) if r8 \u003e r0 goto pc+1\n     R0\u003dpkt_end R1\u003dinv56 R6\u003dctx R7\u003dpkt(id\u003d0,off\u003d0,r\u003d32) R8\u003dpkt(id\u003d1,off\u003d32,r\u003d32) R9\u003dpkt(id\u003d1,off\u003d0,r\u003d32) R10\u003dfp\n    14: (71) r1 \u003d *(u8 *)(r9 +16)\n    15: (b7) r7 \u003d 0\n    16: (bf) r0 \u003d r7\n    17: (95) exit\n\n    We need to get a UNKNOWN_VALUE with imm to force id\n    generation so lines 0-5 make r7 a valid packet pointer.\n    We then read two different bytes from the packet and\n    add them to copies of the constructed packet pointer.\n    r8 (line 9) and r9 (line 11) will get the same id of 1,\n    independently.  When either of them is validated (line\n    13) - find_good_pkt_pointers() will also mark the other\n    as safe.  This leads to access on line 14 being mistakenly\n    considered safe.\n\n    Fixes: 969bf05eb3ce (\"bpf: direct packet access\")\n    Signed-off-by: Jakub Kicinski \u003cjakub.kicinski@netronome.com\u003e\n    Acked-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Acked-by: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit a3275824d8a8398b1e7930f7119a465fb5686992\nAuthor: Brenden Blanco \u003cbblanco@plumgrid.com\u003e\nDate:   Tue Jul 19 12:16:56 2016 -0700\n\n    bpf: enable direct packet data write for xdp progs\n\n    For forwarding to be effective, XDP programs should be allowed to\n    rewrite packet data.\n\n    This requires that the drivers supporting XDP must all map the packet\n    memory as TODEVICE or BIDIRECTIONAL before invoking the program.\n\n    Signed-off-by: Brenden Blanco \u003cbblanco@plumgrid.com\u003e\n    Acked-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 38333684af81f8d4577718edfcd7efcc5920fd90\nAuthor: Brenden Blanco \u003cbblanco@plumgrid.com\u003e\nDate:   Tue Jul 19 12:16:47 2016 -0700\n\n    bpf: add XDP prog type for early driver filter\n\n    Add a new bpf prog type that is intended to run in early stages of the\n    packet rx path. Only minimal packet metadata will be available, hence a\n    new context type, struct xdp_md, is exposed to userspace. So far only\n    expose the packet start and end pointers, and only in read mode.\n\n    An XDP program must return one of the well known enum values, all other\n    return codes are reserved for future use. Unfortunately, this\n    restriction is hard to enforce at verification time, so take the\n    approach of warning at runtime when such programs are encountered. Out\n    of bounds return codes should alias to XDP_ABORTED.\n\n    Signed-off-by: Brenden Blanco \u003cbblanco@plumgrid.com\u003e\n    Acked-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 6e67961b0098ae6035d1276840fb82d6743a4a17\nAuthor: Alexei Starovoitov \u003cast@fb.com\u003e\nDate:   Thu May 5 19:49:12 2016 -0700\n\n    bpf: wire in data and data_end for cls_act_bpf\n\n    allow cls_bpf and act_bpf programs access skb-\u003edata and skb-\u003edata_end pointers.\n    The bpf helpers that change skb-\u003edata need to update data_end pointer as well.\n    The verifier checks that programs always reload data, data_end pointers\n    after calls to such bpf helpers.\n    We cannot add \u0027data_end\u0027 pointer to struct qdisc_skb_cb directly,\n    since it\u0027s embedded as-is by infiniband ipoib, so wrapper struct is needed.\n\n    Signed-off-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Acked-by: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 7debcb53f327f3ee2c94b9ef33e0508228b1c4c5\nAuthor: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\nDate:   Wed Jan 6 22:32:16 2016 +0100\n\n    bpf: cleanup bpf_prog_run_{save,clear}_cb helpers\n\n    Move the details behind the cb[] access into a small helper to decouple\n    and make them generic for bpf_prog_run_save_cb()/bpf_prog_run_clear_cb()\n    that was introduced via commit ff936a04e5f2 (\"bpf: fix cb access in socket\n    filter programs\"). Also add a comment to better clarify what is done in\n    bpf_skb_cb().\n\n    Signed-off-by: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Acked-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 7e22e35fb420d76dc53f27a264982f54dcb54fd2\nAuthor: Brenden Blanco \u003cbblanco@plumgrid.com\u003e\nDate:   Tue Jul 19 12:16:46 2016 -0700\n\n    bpf: add bpf_prog_add api for bulk prog refcnt\n\n    A subsystem may need to store many copies of a bpf program, each\n    deserving its own reference. Rather than requiring the caller to loop\n    one by one (with possible mid-loop failure), add a bulk bpf_prog_add\n    api.\n\n    Signed-off-by: Brenden Blanco \u003cbblanco@plumgrid.com\u003e\n    Acked-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 72b21d50a9b6decc9405a25c1583b76cdcf7a096\nAuthor: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\nDate:   Sat Jul 16 01:15:55 2016 +0200\n\n    bpf: bpf_event_entry_gen\u0027s alloc needs to be in atomic context\n\n    Should have been obvious, only called from bpf() syscall via map_update_elem()\n    that calls bpf_fd_array_map_update_elem() under RCU read lock and thus this\n    must also be in GFP_ATOMIC, of course.\n\n    Fixes: 3b1efb196eee (\"bpf, maps: flush own entries on perf map release\")\n    Signed-off-by: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Acked-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 6c1aaecc413f00797fbb963a65df8a9ce1477164\nAuthor: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\nDate:   Thu Jul 14 18:08:05 2016 +0200\n\n    bpf: avoid stack copy and use skb ctx for event output\n\n    This work addresses a couple of issues bpf_skb_event_output()\n    helper currently has: i) We need two copies instead of just a\n    single one for the skb data when it should be part of a sample.\n    The data can be non-linear and thus needs to be extracted via\n    bpf_skb_load_bytes() helper first, and then copied once again\n    into the ring buffer slot. ii) Since bpf_skb_load_bytes()\n    currently needs to be used first, the helper needs to see a\n    constant size on the passed stack buffer to make sure BPF\n    verifier can do sanity checks on it during verification time.\n    Thus, just passing skb-\u003elen (or any other non-constant value)\n    wouldn\u0027t work, but changing bpf_skb_load_bytes() is also not\n    the proper solution, since the two copies are generally still\n    needed. iii) bpf_skb_load_bytes() is just for rather small\n    buffers like headers, since they need to sit on the limited\n    BPF stack anyway. Instead of working around in bpf_skb_load_bytes(),\n    this work improves the bpf_skb_event_output() helper to address\n    all 3 at once.\n\n    We can make use of the passed in skb context that we have in\n    the helper anyway, and use some of the reserved flag bits as\n    a length argument. The helper will use the new __output_custom()\n    facility from perf side with bpf_skb_copy() as callback helper\n    to walk and extract the data. It will pass the data for setup\n    to bpf_event_output(), which generates and pushes the raw record\n    with an additional frag part. The linear data used in the first\n    frag of the record serves as programmatically defined meta data\n    passed along with the appended sample.\n\n    Signed-off-by: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Acked-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 59edbb6a3fb74ac235bfabecaa52ae5bdbac062a\nAuthor: Paul Gortmaker \u003cpaul.gortmaker@windriver.com\u003e\nDate:   Mon Jul 11 12:51:01 2016 -0400\n\n    bpf: make inode code explicitly non-modular\n\n    The Kconfig currently controlling compilation of this code is:\n\n    init/Kconfig:config BPF_SYSCALL\n    init/Kconfig:   bool \"Enable bpf() system call\"\n\n    ...meaning that it currently is not being built as a module by anyone.\n\n    Lets remove the couple traces of modular infrastructure use, so that\n    when reading the driver there is no doubt it is builtin-only.\n\n    Note that MODULE_ALIAS is a no-op for non-modular code.\n\n    We replace module.h with init.h since the file does use __init.\n\n    Cc: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Cc: netdev@vger.kernel.org\n    Signed-off-by: Paul Gortmaker \u003cpaul.gortmaker@windriver.com\u003e\n    Acked-by: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 089d7966b53b0fed60e2e60929fb5965c818fb6c\nAuthor: Martin KaFai Lau \u003ckafai@fb.com\u003e\nDate:   Thu Jun 30 10:28:44 2016 -0700\n\n    cgroup: bpf: Add bpf_skb_in_cgroup_proto\n\n    Adds a bpf helper, bpf_skb_in_cgroup, to decide if a skb-\u003esk\n    belongs to a descendant of a cgroup2.  It is similar to the\n    feature added in netfilter:\n    commit c38c4597e4bf (\"netfilter: implement xt_cgroup cgroup2 path match\")\n\n    The user is expected to populate a BPF_MAP_TYPE_CGROUP_ARRAY\n    which will be used by the bpf_skb_in_cgroup.\n\n    Modifications to the bpf verifier is to ensure BPF_MAP_TYPE_CGROUP_ARRAY\n    and bpf_skb_in_cgroup() are always used together.\n\n    Signed-off-by: Martin KaFai Lau \u003ckafai@fb.com\u003e\n    Cc: Alexei Starovoitov \u003cast@fb.com\u003e\n    Cc: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Cc: Tejun Heo \u003ctj@kernel.org\u003e\n    Acked-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit a88dd08f28fa22d29407cf9a9d0fbc49f1c5c3f0\nAuthor: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\nDate:   Tue Jun 28 12:18:28 2016 +0200\n\n    bpf: add bpf_skb_change_type helper\n\n    This work adds a helper for changing skb-\u003epkt_type in a controlled way.\n    We only allow a subset of possible values and can extend that in future\n    should other use cases come up. Doing this as a helper has the advantage\n    that errors can be handeled gracefully and thus helper kept extensible.\n\n    It\u0027s a write counterpart to pkt_type member we can already read from\n    struct __sk_buff context. Major use case is to change incoming skbs to\n    PACKET_HOST in a programmatic way instead of having to recirculate via\n    redirect(..., BPF_F_INGRESS), for example.\n\n    Signed-off-by: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Acked-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 1d8b2c9550246c109da3db6af52a341af0e3f7ca\nAuthor: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\nDate:   Tue Jun 28 12:18:27 2016 +0200\n\n    bpf: add bpf_skb_change_proto helper\n\n    This patch adds a minimal helper for doing the groundwork of changing\n    the skb-\u003eprotocol in a controlled way. Currently supported is v4 to\n    v6 and vice versa transitions, which allows f.e. for a minimal, static\n    nat64 implementation where applications in containers that still\n    require IPv4 can be transparently operated in an IPv6-only environment.\n    For example, host facing veth of the container can transparently do\n    the transitions in a programmatic way with the help of clsact qdisc\n    and cls_bpf.\n\n    Idea is to separate concerns for keeping complexity of the helper\n    lower, which means that the programs utilize bpf_skb_change_proto(),\n    bpf_skb_store_bytes() and bpf_lX_csum_replace() to get the job done,\n    instead of doing everything in a single helper (and thus partially\n    duplicating helper functionality). Also, bpf_skb_change_proto()\n    shouldn\u0027t need to deal with raw packet data as this is done by other\n    helpers.\n\n    bpf_skb_proto_6_to_4() and bpf_skb_proto_4_to_6() unclone the skb to\n    operate on a private one, push or pop additionally required header\n    space and migrate the gso/gro meta data from the shared info. We do\n    mark the gso type as dodgy so that headers are checked and segs\n    recalculated by the gso/gro engine. The gso_size target is adapted\n    as well. The flags argument added is currently reserved and can be\n    used for future extensions.\n\n    Signed-off-by: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Acked-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 68f2bce405632a5b582db4a78cfa152c3f072063\nAuthor: Martin KaFai Lau \u003ckafai@fb.com\u003e\nDate:   Thu Jun 30 10:28:43 2016 -0700\n\n    cgroup: bpf: Add BPF_MAP_TYPE_CGROUP_ARRAY\n\n    Add a BPF_MAP_TYPE_CGROUP_ARRAY and its bpf_map_ops\u0027s implementations.\n    To update an element, the caller is expected to obtain a cgroup2 backed\n    fd by open(cgroup2_dir) and then update the array with that fd.\n\n    Signed-off-by: Martin KaFai Lau \u003ckafai@fb.com\u003e\n    Cc: Alexei Starovoitov \u003cast@fb.com\u003e\n    Cc: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Cc: Tejun Heo \u003ctj@kernel.org\u003e\n    Acked-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit c5df69e626fe49a594178ba1ddc919c27d4ed765\nAuthor: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\nDate:   Thu Jun 30 17:24:44 2016 +0200\n\n    bpf: refactor bpf_prog_get and type check into helper\n\n    Since bpf_prog_get() and program type check is used in a couple of places,\n    refactor this into a small helper function that we can make use of. Since\n    the non RO prog-\u003eaux part is not used in performance critical paths and a\n    program destruction via RCU is rather very unlikley when doing the put, we\n    shouldn\u0027t have an issue just doing the bpf_prog_get() + prog-\u003etype !\u003d type\n    check, but actually not taking the ref at all (due to being in fdget() /\n    fdput() section of the bpf fd) is even cleaner and makes the diff smaller\n    as well, so just go for that. Callsites are changed to make use of the new\n    helper where possible.\n\n    Signed-off-by: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Acked-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 630ac0a763cff191499103976d8434c40d5b4bda\nAuthor: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\nDate:   Thu Jun 30 17:24:43 2016 +0200\n\n    bpf: generally move prog destruction to RCU deferral\n\n    Jann Horn reported following analysis that could potentially result\n    in a very hard to trigger (if not impossible) UAF race, to quote his\n    event timeline:\n\n     - Set up a process with threads T1, T2 and T3\n     - Let T1 set up a socket filter F1 that invokes another filter F2\n       through a BPF map [tail call]\n     - Let T1 trigger the socket filter via a unix domain socket write,\n       don\u0027t wait for completion\n     - Let T2 call PERF_EVENT_IOC_SET_BPF with F2, don\u0027t wait for completion\n     - Now T2 should be behind bpf_prog_get(), but before bpf_prog_put()\n     - Let T3 close the file descriptor for F2, dropping the reference\n       count of F2 to 2\n     - At this point, T1 should have looked up F2 from the map, but not\n       finished executing it\n     - Let T3 remove F2 from the BPF map, dropping the reference count of\n       F2 to 1\n     - Now T2 should call bpf_prog_put() (wrong BPF program type), dropping\n       the reference count of F2 to 0 and scheduling bpf_prog_free_deferred()\n       via schedule_work()\n     - At this point, the BPF program could be freed\n     - BPF execution is still running in a freed BPF program\n\n    While at PERF_EVENT_IOC_SET_BPF time it\u0027s only guaranteed that the perf\n    event fd we\u0027re doing the syscall on doesn\u0027t disappear from underneath us\n    for whole syscall time, it may not be the case for the bpf fd used as\n    an argument only after we did the put. It needs to be a valid fd pointing\n    to a BPF program at the time of the call to make the bpf_prog_get() and\n    while T2 gets preempted, F2 must have dropped reference to 1 on the other\n    CPU. The fput() from the close() in T3 should also add additionally delay\n    to the reference drop via exit_task_work() when bpf_prog_release() gets\n    called as well as scheduling bpf_prog_free_deferred().\n\n    That said, it makes nevertheless sense to move the BPF prog destruction\n    generally after RCU grace period to guarantee that such scenario above,\n    but also others as recently fixed in ceb56070359b (\"bpf, perf: delay release\n    of BPF prog after grace period\") with regards to tail calls won\u0027t happen.\n    Integrating bpf_prog_free_deferred() directly into the RCU callback is\n    not allowed since the invocation might happen from either softirq or\n    process context, so we\u0027re not permitted to block. Reviewing all bpf_prog_put()\n    invocations from eBPF side (note, cBPF -\u003e eBPF progs don\u0027t use this for\n    their destruction) with call_rcu() look good to me.\n\n    Since we don\u0027t know whether at the time of attaching the program, we\u0027re\n    already part of a tail call map, we need to use RCU variant. However, due\n    to this, there won\u0027t be severely more stress on the RCU callback queue:\n    situations with above bpf_prog_get() and bpf_prog_put() combo in practice\n    normally won\u0027t lead to releases, but even if they would, enough effort/\n    cycles have to be put into loading a BPF program into the kernel already.\n\n    Reported-by: Jann Horn \u003cjannh@google.com\u003e\n    Signed-off-by: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Acked-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit d61c3ada991b34d24cf191a2426b8300c848e945\nAuthor: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\nDate:   Tue Jun 28 12:18:26 2016 +0200\n\n    bpf: don\u0027t use raw processor id in generic helper\n\n    Use smp_processor_id() for the generic helper bpf_get_smp_processor_id()\n    instead of the raw variant. This allows for preemption checks when we\n    have DEBUG_PREEMPT, and otherwise uses the raw variant anyway. We only\n    need to keep the raw variant for socket filters, but we can reuse the\n    helper that is already there from cBPF side.\n\n    Signed-off-by: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Acked-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 362d8463e5702958ef19b610c3a2b5bd51794a27\nAuthor: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\nDate:   Tue Jun 28 12:18:23 2016 +0200\n\n    bpf: minor cleanups on fd maps and helpers\n\n    Some minor cleanups: i) Remove the unlikely() from fd array map lookups\n    and let the CPU branch predictor do its job, scenarios where there is not\n    always a map entry are very well valid. ii) Move the attribute type check\n    in the bpf_perf_event_read() helper a bit earlier so it\u0027s consistent wrt\n    checks with bpf_perf_event_output() helper as well. iii) remove some\n    comments that are self-documenting in kprobe_prog_is_valid_access() and\n    therefore make it consistent to tp_prog_is_valid_access() as well.\n\n    Signed-off-by: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Acked-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit d64100bfde0a60b88c30c9d9cf3e233378da7228\nAuthor: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\nDate:   Wed Jun 15 22:47:14 2016 +0200\n\n    bpf, maps: flush own entries on perf map release\n\n    The behavior of perf event arrays are quite different from all\n    others as they are tightly coupled to perf event fds, f.e. shown\n    recently by commit e03e7ee34fdd (\"perf/bpf: Convert perf_event_array\n    to use struct file\") to make refcounting on perf event more robust.\n    A remaining issue that the current code still has is that since\n    additions to the perf event array take a reference on the struct\n    file via perf_event_get() and are only released via fput() (that\n    cleans up the perf event eventually via perf_event_release_kernel())\n    when the element is either manually removed from the map from user\n    space or automatically when the last reference on the perf event\n    map is dropped. However, this leads us to dangling struct file\u0027s\n    when the map gets pinned after the application owning the perf\n    event descriptor exits, and since the struct file reference will\n    in such case only be manually dropped or via pinned file removal,\n    it leads to the perf event living longer than necessary, consuming\n    needlessly resources for that time.\n\n    Relations between perf event fds and bpf perf event map fds can be\n    rather complex. F.e. maps can act as demuxers among different perf\n    event fds that can possibly be owned by different threads and based\n    on the index selection from the program, events get dispatched to\n    one of the per-cpu fd endpoints. One perf event fd (or, rather a\n    per-cpu set of them) can also live in multiple perf event maps at\n    the same time, listening for events. Also, another requirement is\n    that perf event fds can get closed from application side after they\n    have been attached to the perf event map, so that on exit perf event\n    map will take care of dropping their references eventually. Likewise,\n    when such maps are pinned, the intended behavior is that a user\n    application does bpf_obj_get(), puts its fds in there and on exit\n    when fd is released, they are dropped from the map again, so the map\n    acts rather as connector endpoint. This also makes perf event maps\n    inherently different from program arrays as described in more detail\n    in commit c9da161c6517 (\"bpf: fix clearing on persistent program\n    array maps\").\n\n    To tackle this, map entries are marked by the map struct file that\n    added the element to the map. And when the last reference to that map\n    struct file is released from user space, then the tracked entries\n    are purged from the map. This is okay, because new map struct files\n    instances resp. frontends to the anon inode are provided via\n    bpf_map_new_fd() that is called when we invoke bpf_obj_get_user()\n    for retrieving a pinned map, but also when an initial instance is\n    created via map_create(). The rest is resolved by the vfs layer\n    automatically for us by keeping reference count on the map\u0027s struct\n    file. Any concurrent updates on the map slot are fine as well, it\n    just means that perf_event_fd_array_release() needs to delete less\n    of its own entires.\n\n    Signed-off-by: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Acked-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 4fdb9c175239526ba8e7997049c5359df25a4295\nAuthor: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\nDate:   Wed Jun 15 22:47:13 2016 +0200\n\n    bpf, maps: extend map_fd_get_ptr arguments\n\n    This patch extends map_fd_get_ptr() callback that is used by fd array\n    maps, so that struct file pointer from the related map can be passed\n    in. It\u0027s safe to remove map_update_elem() callback for the two maps since\n    this is only allowed from syscall side, but not from eBPF programs for these\n    two map types. Like in per-cpu map case, bpf_fd_array_map_update_elem()\n    needs to be called directly here due to the extra argument.\n\n    Signed-off-by: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Acked-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 9275a36a33bd1f924b07196614651a8f83fc96f6\nAuthor: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\nDate:   Wed Jun 15 22:47:12 2016 +0200\n\n    bpf, maps: add release callback\n\n    Add a release callback for maps that is invoked when the last\n    reference to its struct file is gone and the struct file about\n    to be released by vfs. The handler will be used by fd array maps.\n\n    Signed-off-by: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Acked-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit b6d5778f26b112d6df6d7a5b8e76799bf79dccb6\nAuthor: Alexei Starovoitov \u003cast@fb.com\u003e\nDate:   Wed Jun 15 18:25:38 2016 -0700\n\n    bpf: fix matching of data/data_end in verifier\n\n    The ctx structure passed into bpf programs is different depending on bpf\n    program type. The verifier incorrectly marked ctx-\u003edata and ctx-\u003edata_end\n    access based on ctx offset only. That caused loads in tracing programs\n    int bpf_prog(struct pt_regs *ctx) { .. ctx-\u003eax .. }\n    to be incorrectly marked as PTR_TO_PACKET which later caused verifier\n    to reject the program that was actually valid in tracing context.\n    Fix this by doing program type specific matching of ctx offsets.\n\n    Fixes: 969bf05eb3ce (\"bpf: direct packet access\")\n    Reported-by: Sasha Goldshtein \u003cgoldshtn@gmail.com\u003e\n    Signed-off-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Acked-by: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 8aad11b3c53363a32c1652ece0c3471b8ab48a67\nAuthor: Arnaldo Carvalho de Melo \u003cacme@redhat.com\u003e\nDate:   Thu Apr 28 13:16:33 2016 -0300\n\n    perf core: Per event callchain limit\n\n    Additionally to being able to control the system wide maximum depth via\n    /proc/sys/kernel/perf_event_max_stack, now we are able to ask for\n    different depths per event, using perf_event_attr.sample_max_stack for\n    that.\n\n    This uses an u16 hole at the end of perf_event_attr, that, when\n    perf_event_attr.sample_type has the PERF_SAMPLE_CALLCHAIN, if\n    sample_max_stack is zero, means use perf_event_max_stack, otherwise\n    it\u0027ll be bounds checked under callchain_mutex.\n\n    Cc: Adrian Hunter \u003cadrian.hunter@intel.com\u003e\n    Cc: Alexander Shishkin \u003calexander.shishkin@linux.intel.com\u003e\n    Cc: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Cc: Brendan Gregg \u003cbrendan.d.gregg@gmail.com\u003e\n    Cc: David Ahern \u003cdsahern@gmail.com\u003e\n    Cc: Frederic Weisbecker \u003cfweisbec@gmail.com\u003e\n    Cc: He Kuang \u003chekuang@huawei.com\u003e\n    Cc: Jiri Olsa \u003cjolsa@redhat.com\u003e\n    Cc: Linus Torvalds \u003ctorvalds@linux-foundation.org\u003e\n    Cc: Masami Hiramatsu \u003cmhiramat@kernel.org\u003e\n    Cc: Milian Wolff \u003cmilian.wolff@kdab.com\u003e\n    Cc: Namhyung Kim \u003cnamhyung@kernel.org\u003e\n    Cc: Peter Zijlstra \u003cpeterz@infradead.org\u003e\n    Cc: Stephane Eranian \u003ceranian@google.com\u003e\n    Cc: Thomas Gleixner \u003ctglx@linutronix.de\u003e\n    Cc: Vince Weaver \u003cvincent.weaver@maine.edu\u003e\n    Cc: Wang Nan \u003cwangnan0@huawei.com\u003e\n    Cc: Zefan Li \u003clizefan@huawei.com\u003e\n    Link: http://lkml.kernel.org/n/tip-kolmn1yo40p7jhswxwrc7rrd@git.kernel.org\n    Signed-off-by: Arnaldo Carvalho de Melo \u003cacme@redhat.com\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 61cc59ac4f6d1d84ae48830ed8b70de3b655a133\nAuthor: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\nDate:   Sun May 22 23:16:18 2016 +0200\n\n    bpf, inode: disallow userns mounts\n\n    Follow-up to commit e27f4a942a0e (\"bpf: Use mount_nodev not mount_ns\n    to mount the bpf filesystem\"), which removes the FS_USERNS_MOUNT flag.\n\n    The original idea was to have a per mountns instance instead of a\n    single global fs instance, but that didn\u0027t work out and we had to\n    switch to mount_nodev() model. The intent of that middle ground was\n    that we avoid users who don\u0027t play nice to create endless instances\n    of bpf fs which are difficult to control and discover from an admin\n    point of view, but at the same time it would have allowed us to be\n    more flexible with regard to namespaces.\n\n    Therefore, since we now did the switch to mount_nodev() as a fix\n    where individual instances are created, we also need to remove userns\n    mount flag along with it to avoid running into mentioned situation.\n    I don\u0027t expect any breakage at this early point in time with removing\n    the flag and we can revisit this later should the requirement for\n    this come up with future users. This and commit e27f4a942a0e have\n    been split to facilitate tracking should any of them run into the\n    unlikely case of causing a regression.\n\n    Fixes: b2197755b263 (\"bpf: add support for persistent maps/progs\")\n    Signed-off-by: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Acked-by: Hannes Frederic Sowa \u003channes@stressinduktion.org\u003e\n    Acked-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 7db5c6a6926a11cc5d4d288d02f6d947ba784d06\nAuthor: Alexei Starovoitov \u003cast@fb.com\u003e\nDate:   Thu May 19 18:17:14 2016 -0700\n\n    bpf: teach verifier to recognize imm +\u003d ptr pattern\n\n    Humans don\u0027t write C code like:\n      u8 *ptr \u003d skb-\u003edata;\n      int imm \u003d 4;\n      imm +\u003d ptr;\n    but from llvm backend point of view \u0027imm\u0027 and \u0027ptr\u0027 are registers and\n    imm +\u003d ptr may be preferred vs ptr +\u003d imm depending which register value\n    will be used further in the code, while verifier can only recognize ptr +\u003d imm.\n    That caused small unrelated changes in the C code of the bpf program to\n    trigger rejection by the verifier. Therefore teach the verifier to recognize\n    both ptr +\u003d imm and imm +\u003d ptr.\n    For example:\n    when R6\u003dpkt(id\u003d0,off\u003d0,r\u003d62) R7\u003dimm22\n    after r7 +\u003d r6 instruction\n    will be R6\u003dpkt(id\u003d0,off\u003d0,r\u003d62) R7\u003dpkt(id\u003d0,off\u003d22,r\u003d62)\n\n    Fixes: 969bf05eb3ce (\"bpf: direct packet access\")\n    Signed-off-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Acked-by: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 6fd0e1d829706abebbe1ebdaef8b1cbc6b4c24b8\nAuthor: Alexei Starovoitov \u003cast@fb.com\u003e\nDate:   Thu May 19 18:17:13 2016 -0700\n\n    bpf: support decreasing order in direct packet access\n\n    when packet headers are accessed in \u0027decreasing\u0027 order (like TCP port\n    may be fetched before the program reads IP src) the llvm may generate\n    the following code:\n    [...]                // R7\u003dpkt(id\u003d0,off\u003d22,r\u003d70)\n    r2 \u003d *(u32 *)(r7 +0) // good access\n    [...]\n    r7 +\u003d 40             // R7\u003dpkt(id\u003d0,off\u003d62,r\u003d70)\n    r8 \u003d *(u32 *)(r7 +0) // good access\n    [...]\n    r1 \u003d *(u32 *)(r7 -20) // this one will fail though it\u0027s within a safe range\n                          // it\u0027s doing *(u32*)(skb-\u003edata + 42)\n    Fix verifier to recognize such code pattern\n\n    Alos turned out that \u0027off \u003e range\u0027 condition is not a verifier bug.\n    It\u0027s a buggy program that may do something like:\n    if (ptr + 50 \u003e data_end)\n      return 0;\n    ptr +\u003d 60;\n    *(u32*)ptr;\n    in such case emit\n    \"invalid access to packet, off\u003d0 size\u003d4, R1(id\u003d0,off\u003d60,r\u003d50)\" error message,\n    so all information is available for the program author to fix the program.\n\n    Fixes: 969bf05eb3ce (\"bpf: direct packet access\")\n    Signed-off-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Acked-by: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 9ff7a276a4eade9c51ad5c8b1432e950e11f86fd\nAuthor: Eric W. Biederman \u003cebiederm@xmission.com\u003e\nDate:   Fri May 20 17:22:48 2016 -0500\n\n    bpf: Use mount_nodev not mount_ns to mount the bpf filesystem\n\n    While reviewing the filesystems that set FS_USERNS_MOUNT I spotted the\n    bpf filesystem.  Looking at the code I saw a broken usage of mount_ns\n    with current-\u003ensproxy-\u003emnt_ns. As the code does not acquire a\n    reference to the mount namespace it can not possibly be correct to\n    store the mount namespace on the superblock as it does.\n\n    Replace mount_ns with mount_nodev so that each mount of the bpf\n    filesystem returns a distinct instance, and the code is not buggy.\n\n    In discussion with Hannes Frederic Sowa it was reported that the use\n    of mount_ns was an attempt to have one bpf instance per mount\n    namespace, in an attempt to keep resources that pin resources from\n    hiding.  That intent simply does not work, the vfs is not built to\n    allow that kind of behavior.  Which means that the bpf filesystem\n    really is buggy both semantically and in it\u0027s implemenation as it does\n    not nor can it implement the original intent.\n\n    This change is userspace visible, but my experience with similar\n    filesystems leads me to believe nothing will break with a model of each\n    mount of the bpf filesystem is distinct from all others.\n\n    Fixes: b2197755b263 (\"bpf: add support for persistent maps/progs\")\n    Cc: Hannes Frederic Sowa \u003channes@stressinduktion.org\u003e\n    Acked-by: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Signed-off-by: \"Eric W. Biederman\" \u003cebiederm@xmission.com\u003e\n    Acked-by: Hannes Frederic Sowa \u003channes@stressinduktion.org\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit b45ed4de08972ccce7528e87d6fc77e43b8dc783\nAuthor: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\nDate:   Wed May 18 14:14:28 2016 +0200\n\n    bpf: rather use get_random_int for randomizations\n\n    Start address randomization and blinding in BPF currently use\n    prandom_u32(). prandom_u32() values are not exposed to unpriviledged\n    user space to my knowledge, but given other kernel facilities such as\n    ASLR, stack canaries, etc make use of stronger get_random_int(), we\n    better make use of it here as well given blinding requests successively\n    new random values. get_random_int() has minimal entropy pool depletion,\n    is not cryptographically secure, but doesn\u0027t need to be for our use\n    cases here.\n\n    Suggested-by: Hannes Frederic Sowa \u003channes@stressinduktion.org\u003e\n    Signed-off-by: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Acked-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit c033a4e3da3f633dcdd5c817e2a316b8ced52d44\nAuthor: Arnaldo Carvalho de Melo \u003cacme@redhat.com\u003e\nDate:   Thu Apr 28 12:30:53 2016 -0300\n\n    perf core: Pass max stack as a perf_callchain_entry context\n\n    This makes perf_callchain_{user,kernel}() receive the max stack\n    as context for the perf_callchain_entry, instead of accessing\n    the global sysctl_perf_event_max_stack.\n\n    Cc: Adrian Hunter \u003cadrian.hunter@intel.com\u003e\n    Cc: Alexander Shishkin \u003calexander.shishkin@linux.intel.com\u003e\n    Cc: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Cc: Brendan Gregg \u003cbrendan.d.gregg@gmail.com\u003e\n    Cc: David Ahern \u003cdsahern@gmail.com\u003e\n    Cc: Frederic Weisbecker \u003cfweisbec@gmail.com\u003e\n    Cc: He Kuang \u003chekuang@huawei.com\u003e\n    Cc: Jiri Olsa \u003cjolsa@redhat.com\u003e\n    Cc: Linus Torvalds \u003ctorvalds@linux-foundation.org\u003e\n    Cc: Masami Hiramatsu \u003cmhiramat@kernel.org\u003e\n    Cc: Milian Wolff \u003cmilian.wolff@kdab.com\u003e\n    Cc: Namhyung Kim \u003cnamhyung@kernel.org\u003e\n    Cc: Peter Zijlstra \u003cpeterz@infradead.org\u003e\n    Cc: Stephane Eranian \u003ceranian@google.com\u003e\n    Cc: Thomas Gleixner \u003ctglx@linutronix.de\u003e\n    Cc: Vince Weaver \u003cvincent.weaver@maine.edu\u003e\n    Cc: Wang Nan \u003cwangnan0@huawei.com\u003e\n    Cc: Zefan Li \u003clizefan@huawei.com\u003e\n    Link: http://lkml.kernel.org/n/tip-kolmn1yo40p7jhswxwrc7rrd@git.kernel.org\n    Signed-off-by: Arnaldo Carvalho de Melo \u003cacme@redhat.com\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit cabf4cf98c23ad1e4edf583e130e99387993ed3a\nAuthor: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\nDate:   Fri May 13 19:08:32 2016 +0200\n\n    bpf: add generic constant blinding for use in jits\n\n    This work adds a generic facility for use from eBPF JIT compilers\n    that allows for further hardening of JIT generated images through\n    blinding constants. In response to the original work on BPF JIT\n    spraying published by Keegan McAllister [1], most BPF JITs were\n    changed to make images read-only and start at a randomized offset\n    in the page, where the rest was filled with trap instructions. We\n    have this nowadays in x86, arm, arm64 and s390 JIT compilers.\n    Additionally, later work also made eBPF interpreter images read\n    only for kernels supporting DEBUG_SET_MODULE_RONX, that is, x86,\n    arm, arm64 and s390 archs as well currently. This is done by\n    default for mentioned JITs when JITing is enabled. Furthermore,\n    we had a generic and configurable constant blinding facility on our\n    todo for quite some time now to further make spraying harder, and\n    first implementation since around netconf 2016.\n\n    We found that for systems where untrusted users can load cBPF/eBPF\n    code where JIT is enabled, start offset randomization helps a bit\n    to make jumps into crafted payload harder, but in case where larger\n    programs that cross page boundary are injected, we again have some\n    part of the program opcodes at a page start offset. With improved\n    guessing and more reliable payload injection, chances can increase\n    to jump into such payload. Elena Reshetova recently wrote a test\n    case for it [2, 3]. Moreover, eBPF comes with 64 bit constants, which\n    can leave some more room for payloads. Note that for all this,\n    additional bugs in the kernel are still required to make the jump\n    (and of course to guess right, to not jump into a trap) and naturally\n    the JIT must be enabled, which is disabled by default.\n\n    For helping mitigation, the general idea is to provide an option\n    bpf_jit_harden that admins can tweak along with bpf_jit_enable, so\n    that for cases where JIT should be enabled for performance reasons,\n    the generated image can be further hardened with blinding constants\n    for unpriviledged users (bpf_jit_harden \u003d\u003d 1), with trading off\n    performance for these, but not for privileged ones. We also added\n    the option of blinding for all users (bpf_jit_harden \u003d\u003d 2), which\n    is quite helpful for testing f.e. with test_bpf.ko. There are no\n    further e.g. hardening levels of bpf_jit_harden switch intended,\n    rationale is to have it dead simple to use as on/off. Since this\n    functionality would need to be duplicated over and over for JIT\n    compilers to use, which are already complex enough, we provide a\n    generic eBPF byte-code level based blinding implementation, which is\n    then just transparently JITed. JIT compilers need to make only a few\n    changes to integrate this facility and can be migrated one by one.\n\n    This option is for eBPF JITs and will be used in x86, arm64, s390\n    without too much effort, and soon ppc64 JITs, thus that native eBPF\n    can be blinded as well as cBPF to eBPF migrations, so that both can\n    be covered with a single implementation. The rule for JITs is that\n    bpf_jit_blind_constants() must be called from bpf_int_jit_compile(),\n    and in case blinding is disabled, we follow normally with JITing the\n    passed program. In case blinding is enabled and we fail during the\n    process of blinding itself, we must return with the interpreter.\n    Similarly, in case the JITing process after the blinding failed, we\n    return normally to the interpreter with the non-blinded code. Meaning,\n    interpreter doesn\u0027t change in any way and operates on eBPF code as\n    usual. For doing this pre-JIT blinding step, we need to make use of\n    a helper/auxiliary register, here BPF_REG_AX. This is strictly internal\n    to the JIT and not in any way part of the eBPF architecture. Just like\n    in the same way as JITs internally make use of some helper registers\n    when emitting code, only that here the helper register is one\n    abstraction level higher in eBPF bytecode, but nevertheless in JIT\n    phase. That helper register is needed since f.e. manually written\n    program can issue loads to all registers of eBPF architecture.\n\n    The core concept with the additional register is: blind out all 32\n    and 64 bit constants by converting BPF_K based instructions into a\n    small sequence from K_VAL into ((RND ^ K_VAL) ^ RND). Therefore, this\n    is transformed into: BPF_REG_AX :\u003d (RND ^ K_VAL), BPF_REG_AX ^\u003d RND,\n    and REG \u003cOP\u003e BPF_REG_AX, so actual operation on the target register\n    is translated from BPF_K into BPF_X one that is operating on\n    BPF_REG_AX\u0027s content. During rewriting phase when blinding, RND is\n    newly generated via prandom_u32() for each processed instruction.\n    64 bit loads are split into two 32 bit loads to make translation and\n    patching not too complex. Only basic thing required by JITs is to\n    call the helper bpf_jit_blind_constants()/bpf_jit_prog_release_other()\n    pair, and to map BPF_REG_AX into an unused register.\n\n    Small bpf_jit_disasm extract from [2] when applied to x86 JIT:\n\n    echo 0 \u003e /proc/sys/net/core/bpf_jit_harden\n\n      ffffffffa034f5e9 + \u003cx\u003e:\n      [...]\n      39:   mov    $0xa8909090,%eax\n      3e:   mov    $0xa8909090,%eax\n      43:   mov    $0xa8ff3148,%eax\n      48:   mov    $0xa89081b4,%eax\n      4d:   mov    $0xa8900bb0,%eax\n      52:   mov    $0xa810e0c1,%eax\n      57:   mov    $0xa8908eb4,%eax\n      5c:   mov    $0xa89020b0,%eax\n      [...]\n\n    echo 1 \u003e /proc/sys/net/core/bpf_jit_harden\n\n      ffffffffa034f1e5 + \u003cx\u003e:\n      [...]\n      39:   mov    $0xe1192563,%r10d\n      3f:   xor    $0x4989b5f3,%r10d\n      46:   mov    %r10d,%eax\n      49:   mov    $0xb8296d93,%r10d\n      4f:   xor    $0x10b9fd03,%r10d\n      56:   mov    %r10d,%eax\n      59:   mov    $0x8c381146,%r10d\n      5f:   xor    $0x24c7200e,%r10d\n      66:   mov    %r10d,%eax\n      69:   mov    $0xeb2a830e,%r10d\n      6f:   xor    $0x43ba02ba,%r10d\n      76:   mov    %r10d,%eax\n      79:   mov    $0xd9730af,%r10d\n      7f:   xor    $0xa5073b1f,%r10d\n      86:   mov    %r10d,%eax\n      89:   mov    $0x9a45662b,%r10d\n      8f:   xor    $0x325586ea,%r10d\n      96:   mov    %r10d,%eax\n      [...]\n\n    As can be seen, original constants that carry payload are hidden\n    when enabled, actual operations are transformed from constant-based\n    to register-based ones, making jumps into constants ineffective.\n    Above extract/example uses single BPF load instruction over and\n    over, but of course all instructions with constants are blinded.\n\n    Performance wise, JIT with blinding performs a bit slower than just\n    JIT and faster than interpreter case. This is expected, since we\n    still get all the performance benefits from JITing and in normal\n    use-cases not every single instruction needs to be blinded. Summing\n    up all 296 test cases averaged over multiple runs from test_bpf.ko\n    suite, interpreter was 55% slower than JIT only and JIT with blinding\n    was 8% slower than JIT only. Since there are also some extremes in\n    the test suite, I expect for ordinary workloads that the performance\n    for the JIT with blinding case is even closer to JIT only case,\n    f.e. nmap test case from suite has averaged timings in ns 29 (JIT),\n    35 (+ blinding), and 151 (interpreter).\n\n    BPF test suite, seccomp test suite, eBPF sample code and various\n    bigger networking eBPF programs have been tested with this and were\n    running fine. For testing purposes, I also adapted interpreter and\n    redirected blinded eBPF image to interpreter and also here all tests\n    pass.\n\n      [1] http://mainisusuallyafunction.blogspot.com/2012/11/attacking-hardened-linux-systems-with.html\n      [2] https://github.com/01org/jit-spray-poc-for-ksp/\n      [3] http://www.openwall.com/lists/kernel-hardening/2016/05/03/5\n\n    Signed-off-by: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Reviewed-by: Elena Reshetova \u003celena.reshetova@intel.com\u003e\n    Acked-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 6463cc096959ff8370d854ac7f0ce1572ecffe38\nAuthor: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\nDate:   Fri May 13 19:08:27 2016 +0200\n\n    bpf: move bpf_jit_enable declaration\n\n    Move the bpf_jit_enable declaration to the filter.h file where\n    most other core code is declared, also since we\u0027re going to add\n    a second knob there.\n\n    Signed-off-by: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Acked-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit dbf9d818c269f7d5a11c7a501ced449e33f74d9f\nAuthor: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\nDate:   Sat Jun 4 20:50:59 2016 +0200\n\n    bpf, trace: use READ_ONCE for retrieving file ptr\n\n    In bpf_perf_event_read() and bpf_perf_event_output(), we must use\n    READ_ONCE() for fetching the struct file pointer, which could get\n    updated concurrently, so we must prevent the compiler from potential\n    refetching.\n\n    We already do this with tail calls for fetching the related bpf_prog,\n    but not so on stored perf events. Semantics for both are the same\n    with regards to updates.\n\n    Fixes: a43eec304259 (\"bpf: introduce bpf_perf_event_output() helper\")\n    Fixes: 35578d798400 (\"bpf: Implement function bpf_perf_event_read() that get the selected hardware PMU conuter\")\n    Signed-off-by: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Acked-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 0840be553a98e9eb514cc064434c34f9d2709ab8\nAuthor: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\nDate:   Mon Apr 18 21:01:23 2016 +0200\n\n    bpf, trace: add BPF_F_CURRENT_CPU flag for bpf_perf_event_output\n\n    Add a BPF_F_CURRENT_CPU flag to optimize the use-case where user space has\n    per-CPU ring buffers and the eBPF program pushes the data into the current\n    CPU\u0027s ring buffer which saves us an extra helper function call in eBPF.\n    Also, make sure to properly reserve the remaining flags which are not used.\n\n    Signed-off-by: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Signed-off-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 0dbb34c4f20b6a60b64ce7580929cc74bb166eec\nAuthor: Arnd Bergmann \u003carnd@arndb.de\u003e\nDate:   Sat Apr 16 22:29:33 2016 +0200\n\n    bpf: avoid warning for wrong pointer cast\n\n    Two new functions in bpf contain a cast from a \u0027u64\u0027 to a\n    pointer. This works on 64-bit architectures but causes a warning\n    on all 32-bit architectures:\n\n    kernel/trace/bpf_trace.c: In function \u0027bpf_perf_event_output_tp\u0027:\n    kernel/trace/bpf_trace.c:350:13: error: cast to pointer from integer of different size [-Werror\u003dint-to-pointer-cast]\n      u64 ctx \u003d *(long *)r1;\n\n    This changes the cast to first convert the u64 argument into a uintptr_t,\n    which is guaranteed to be the same size as a pointer.\n\n    Signed-off-by: Arnd Bergmann \u003carnd@arndb.de\u003e\n    Fixes: 9940d67c93b5 (\"bpf: support bpf_get_stackid() and bpf_perf_event_output() in tracepoint programs\")\n    Acked-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 35b0834b892e0e9a655a65f75d85cdc66e8a2c58\nAuthor: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\nDate:   Fri May 13 19:08:31 2016 +0200\n\n    bpf: prepare bpf_int_jit_compile/bpf_prog_select_runtime apis\n\n    Since the blinding is strictly only called from inside eBPF JITs,\n    we need to change signatures for bpf_int_jit_compile() and\n    bpf_prog_select_runtime() first in order to prepare that the\n    eBPF program we\u0027re dealing with can change underneath. Hence,\n    for call sites, we need to return the latest prog. No functional\n    change in this patch.\n\n    Signed-off-by: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Acked-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 16a7dcfc81a48f8e050813cfd5a4d8f74d7b66fc\nAuthor: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\nDate:   Fri May 13 19:08:30 2016 +0200\n\n    bpf: add bpf_patch_insn_single helper\n\n    Move the functionality to patch instructions out of the verifier\n    code and into the core as the new bpf_patch_insn_single() helper\n    will be needed later on for blinding as well. No changes in\n    functionality.\n\n    Signed-off-by: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Acked-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit a23d1916f3135a6054136459905f46079512057a\nAuthor: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\nDate:   Fri May 13 19:08:26 2016 +0200\n\n    bpf: minor cleanups in ebpf code\n\n    Besides others, remove redundant comments where the code is self\n    documenting enough, and properly indent various bpf_verifier_ops\n    and bpf_prog_type_list declarations. Moreover, remove two exports\n    that actually have no module user.\n\n    Signed-off-by: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Acked-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 4de7f0df019573ffa3dc32bf7b8e7d129df865a1\nAuthor: Alexei Starovoitov \u003cast@fb.com\u003e\nDate:   Thu May 5 19:49:11 2016 -0700\n\n    bpf: improve verifier state equivalence\n\n    since UNKNOWN_VALUE type is weaker than CONST_IMM we can un-teach\n    verifier its recognition of constants in conditional branches\n    without affecting safety.\n    Ex:\n    if (reg \u003d\u003d 123) {\n      .. here verifier was marking reg-\u003etype as CONST_IMM\n         instead keep reg as UNKNOWN_VALUE\n    }\n\n    Two verifier states with UNKNOWN_VALUE are equivalent, whereas\n    CONST_IMM_X !\u003d CONST_IMM_Y, since CONST_IMM is used for stack range\n    verification and other cases.\n    So help search pruning by marking registers as UNKNOWN_VALUE\n    where possible instead of CONST_IMM.\n\n    Signed-off-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Acked-by: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 28d0299e75eb25937f8db1c070149b4ed512f26f\nAuthor: Alexei Starovoitov \u003cast@fb.com\u003e\nDate:   Thu May 5 19:49:10 2016 -0700\n\n    bpf: direct packet access\n\n    Extended BPF carried over two instructions from classic to access\n    packet data: LD_ABS and LD_IND. They\u0027re highly optimized in JITs,\n    but due to their design they have to do length check for every access.\n    When BPF is processing 20M packets per second single LD_ABS after JIT\n    is consuming 3% cpu. Hence the need to optimize it further by amortizing\n    the cost of \u0027off \u003c skb_headlen\u0027 over multiple packet accesses.\n    One option is to introduce two new eBPF instructions LD_ABS_DW and LD_IND_DW\n    with similar usage as skb_header_pointer().\n    The kernel part for interpreter and x64 JIT was implemented in [1], but such\n    new insns behave like old ld_abs and abort the program with \u0027return 0\u0027 if\n    access is beyond linear data. Such hidden control flow is hard to workaround\n    plus changing JITs and rolling out new llvm is incovenient.\n\n    Therefore allow cls_bpf/act_bpf program access skb-\u003edata directly:\n    int bpf_prog(struct __sk_buff *skb)\n    {\n      struct iphdr *ip;\n\n      if (skb-\u003edata + sizeof(struct iphdr) + ETH_HLEN \u003e skb-\u003edata_end)\n          /* packet too small */\n          return 0;\n\n      ip \u003d skb-\u003edata + ETH_HLEN;\n\n      /* access IP header fields with direct loads */\n      if (ip-\u003eversion !\u003d 4 || ip-\u003esaddr \u003d\u003d 0x7f000001)\n          return 1;\n      [...]\n    }\n\n    This solution avoids introduction of new instructions. llvm stays\n    the same and all JITs stay the same, but verifier has to work extra hard\n    to prove safety of the above program.\n\n    For XDP the direct store instructions can be allowed as well.\n\n    The skb-\u003edata is NET_IP_ALIGNED, so for common cases the verifier can check\n    the alignment. The complex packet parsers where packet pointer is adjusted\n    incrementally cannot be tracked for alignment, so allow byte access in such cases\n    and misaligned access on architectures that define efficient_unaligned_access\n\n    [1] https://git.kernel.org/cgit/linux/kernel/git/ast/bpf.git/?h\u003dld_abs_dw\n\n    Signed-off-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Acked-by: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 9dd2b3f13aed7a115ff1a81f296f8603169ff646\nAuthor: Alexei Starovoitov \u003cast@fb.com\u003e\nDate:   Thu May 5 19:49:09 2016 -0700\n\n    bpf: cleanup verifier code\n\n    cleanup verifier code and prepare it for addition of \"pointer to packet\" logic\n\n    Signed-off-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Acked-by: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 9cdf54bc44d0721fbf2784221430be5a2297bd57\nAuthor: Alexei Starovoitov \u003cast@fb.com\u003e\nDate:   Wed Apr 27 18:56:21 2016 -0700\n\n    bpf: fix check_map_func_compatibility logic\n\n    The commit 35578d798400 (\"bpf: Implement function bpf_perf_event_read() that get the selected hardware PMU conuter\")\n    introduced clever way to check bpf_helper\u003c-\u003emap_type compatibility.\n    Later on commit a43eec304259 (\"bpf: introduce bpf_perf_event_output() helper\") adjusted\n    the logic and inadvertently broke it.\n    Get rid of the clever bool compare and go back to two-way check\n    from map and from helper perspective.\n\n    Fixes: a43eec304259 (\"bpf: introduce bpf_perf_event_output() helper\")\n    Reported-by: Jann Horn \u003cjannh@google.com\u003e\n    Signed-off-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Signed-off-by: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 8b6628cac14ddc07a715101123d579566dc74b92\nAuthor: Alexei Starovoitov \u003cast@fb.com\u003e\nDate:   Wed Apr 27 18:56:20 2016 -0700\n\n    bpf: fix refcnt overflow\n\n    On a system with \u003e32Gbyte of phyiscal memory and infinite RLIMIT_MEMLOCK,\n    the malicious application may overflow 32-bit bpf program refcnt.\n    It\u0027s also possible to overflow map refcnt on 1Tb system.\n    Impose 32k hard limit which means that the same bpf program or\n    map cannot be shared by more than 32k processes.\n\n    Fixes: 1be7f75d1668 (\"bpf: enable non-root eBPF programs\")\n    Reported-by: Jann Horn \u003cjannh@google.com\u003e\n    Signed-off-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Acked-by: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 035e484b2a820c5e1702643f27f756e056d1bc86\nAuthor: Arnaldo Carvalho de Melo \u003cacme@redhat.com\u003e\nDate:   Thu Apr 21 12:28:50 2016 -0300\n\n    perf core: Allow setting up max frame stack depth via sysctl\n\n    The default remains 127, which is good for most cases, and not even hit\n    most of the time, but then for some cases, as reported by Brendan, 1024+\n    deep frames are appearing on the radar for things like groovy, ruby.\n\n    And in some workloads putting a _lower_ cap on this may make sense. One\n    that is per event still needs to be put in place tho.\n\n    The new file is:\n\n      # cat /proc/sys/kernel/perf_event_max_stack\n      127\n\n    Chaging it:\n\n      # echo 256 \u003e /proc/sys/kernel/perf_event_max_stack\n      # cat /proc/sys/kernel/perf_event_max_stack\n      256\n\n    But as soon as there is some event using callchains we get:\n\n      # echo 512 \u003e /proc/sys/kernel/perf_event_max_stack\n      -bash: echo: write error: Device or resource busy\n      #\n\n    Because we only allocate the callchain percpu data structures when there\n    is a user, which allows for changing the max easily, its just a matter\n    of having no callchain users at that point.\n\n    Reported-and-Tested-by: Brendan Gregg \u003cbrendan.d.gregg@gmail.com\u003e\n    Reviewed-by: Frederic Weisbecker \u003cfweisbec@gmail.com\u003e\n    Acked-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Acked-by: David Ahern \u003cdsahern@gmail.com\u003e\n    Cc: Adrian Hunter \u003cadrian.hunter@intel.com\u003e\n    Cc: Alexander Shishkin \u003calexander.shishkin@linux.intel.com\u003e\n    Cc: He Kuang \u003chekuang@huawei.com\u003e\n    Cc: Jiri Olsa \u003cjolsa@redhat.com\u003e\n    Cc: Linus Torvalds \u003ctorvalds@linux-foundation.org\u003e\n    Cc: Masami Hiramatsu \u003cmhiramat@kernel.org\u003e\n    Cc: Milian Wolff \u003cmilian.wolff@kdab.com\u003e\n    Cc: Namhyung Kim \u003cnamhyung@kernel.org\u003e\n    Cc: Peter Zijlstra \u003cpeterz@infradead.org\u003e\n    Cc: Stephane Eranian \u003ceranian@google.com\u003e\n    Cc: Thomas Gleixner \u003ctglx@linutronix.de\u003e\n    Cc: Vince Weaver \u003cvincent.weaver@maine.edu\u003e\n    Cc: Wang Nan \u003cwangnan0@huawei.com\u003e\n    Cc: Zefan Li \u003clizefan@huawei.com\u003e\n    Link: http://lkml.kernel.org/r/20160426002928.GB16708@kernel.org\n    Signed-off-by: Arnaldo Carvalho de Melo \u003cacme@redhat.com\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit ded3b8e0b116008682a7b3181b0e252e712a1fed\nAuthor: Alexei Starovoitov \u003cast@fb.com\u003e\nDate:   Wed Feb 17 19:58:57 2016 -0800\n\n    perf: generalize perf_callchain\n\n    . avoid walking the stack when there is no room left in the buffer\n    . generalize get_perf_callchain() to be called from bpf helper\n\n    Signed-off-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 3206acd6ea0a9874b703988c882e2174677d8c5d\nAuthor: Jann Horn \u003cjannh@google.com\u003e\nDate:   Tue Apr 26 22:26:26 2016 +0200\n\n    bpf: fix double-fdput in replace_map_fd_with_map_ptr()\n\n    When bpf(BPF_PROG_LOAD, ...) was invoked with a BPF program whose bytecode\n    references a non-map file descriptor as a map file descriptor, the error\n    handling code called fdput() twice instead of once (in __bpf_map_get() and\n    in replace_map_fd_with_map_ptr()). If the file descriptor table of the\n    current task is shared, this causes f_count to be decremented too much,\n    allowing the struct file to be freed while it is still in use\n    (use-after-free). This can be exploited to gain root privileges by an\n    unprivileged user.\n\n    This bug was introduced in\n    commit 0246e64d9a5f (\"bpf: handle pseudo BPF_LD_IMM64 insn\"), but is only\n    exploitable since\n    commit 1be7f75d1668 (\"bpf: enable non-root eBPF programs\") because\n    previously, CAP_SYS_ADMIN was required to reach the vulnerable code.\n\n    (posted publicly according to request by maintainer)\n\n    Signed-off-by: Jann Horn \u003cjannh@google.com\u003e\n    Signed-off-by: Linus Torvalds \u003ctorvalds@linux-foundation.org\u003e\n    Acked-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Acked-by: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 98783059565f0745f4c8e1e77c3fcee67d60e3ac\nAuthor: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\nDate:   Mon Apr 18 21:01:24 2016 +0200\n\n    bpf: add event output helper for notifications/sampling/logging\n\n    This patch adds a new helper for cls/act programs that can push events\n    to user space applications. For networking, this can be f.e. for sampling,\n    debugging, logging purposes or pushing of arbitrary wake-up events. The\n    idea is similar to a43eec304259 (\"bpf: introduce bpf_perf_event_output()\n    helper\") and 39111695b1b8 (\"samples: bpf: add bpf_perf_event_output example\").\n\n    The eBPF program utilizes a perf event array map that user space populates\n    with fds from perf_event_open(), the eBPF program calls into the helper\n    f.e. as skb_event_output(skb, \u0026my_map, BPF_F_CURRENT_CPU, raw, sizeof(raw))\n    so that the raw data is pushed into the fd f.e. at the map index of the\n    current CPU.\n\n    User space can poll/mmap/etc on this and has a data channel for receiving\n    events that can be post-processed. The nice thing is that since the eBPF\n    program and user space application making use of it are tightly coupled,\n    they can define their own arbitrary raw data format and what/when they\n    want to push.\n\n    While f.e. packet headers could be one part of the meta data that is being\n    pushed, this is not a substitute for things like packet sockets as whole\n    packet is not being pushed and push is only done in a single direction.\n    Intention is more of a generically usable, efficient event pipe to applications.\n    Workflow is that tc can pin the map and applications can attach themselves\n    e.g. after cls/act setup to one or multiple map slots, demuxing is done by\n    the eBPF program.\n\n    Adding this facility is with minimal effort, it reuses the helper\n    introduced in a43eec304259 (\"bpf: introduce bpf_perf_event_output() helper\")\n    and we get its functionality for free by overloading its BPF_FUNC_ identifier\n    for cls/act programs, ctx is currently unused, but will be made use of in\n    future. Example will be added to iproute2\u0027s BPF example files.\n\n    Signed-off-by: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Signed-off-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 5beaebd1d9cc3a168c88072d5922efbee0944958\nAuthor: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\nDate:   Wed Apr 13 00:10:52 2016 +0200\n\n    bpf: convert relevant helper args to ARG_PTR_TO_RAW_STACK\n\n    This patch converts all helpers that can use ARG_PTR_TO_RAW_STACK as argument\n    type. For tc programs this is bpf_skb_load_bytes(), bpf_skb_get_tunnel_key(),\n    bpf_skb_get_tunnel_opt(). For tracing, this optimizes bpf_get_current_comm()\n    and bpf_probe_read(). The check in bpf_skb_load_bytes() for MAX_BPF_STACK can\n    also be removed since the verifier already makes sure we stay within bounds\n    on stack buffers.\n\n    Signed-off-by: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Acked-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit e35f4e25f79f3a553c43e7f690c9d6e167b33745\nAuthor: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\nDate:   Wed Mar 30 00:02:00 2016 +0200\n\n    bpf: make padding in bpf_tunnel_key explicit\n\n    Make the 2 byte padding in struct bpf_tunnel_key between tunnel_ttl\n    and tunnel_label members explicit. No issue has been observed, and\n    gcc/llvm does padding for the old struct already, where tunnel_label\n    was not yet present, so the current code works, but since it\u0027s part\n    of uapi, make sure we don\u0027t introduce holes in structs.\n\n    Therefore, add tunnel_ext that we can use generically in future\n    (f.e. to flag OAM messages for backends, etc). Also add the offset\n    to the compat tests to be sure should some compilers not padd the\n    tail of the old version of bpf_tunnel_key.\n\n    Fixes: 4018ab1875e0 (\"bpf: support flow label for bpf_skb_{set, get}_tunnel_key\")\n    Signed-off-by: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Acked-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 61262beb4b1a111a3f4e63d539c705c8086940fe\nAuthor: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\nDate:   Wed Mar 16 01:42:51 2016 +0100\n\n    ip_tunnels, bpf: define IP_TUNNEL_OPTS_MAX and use it\n\n    eBPF defines this as BPF_TUNLEN_MAX and OVS just uses the hard-coded\n    value inside struct sw_flow_key. Thus, add and use IP_TUNNEL_OPTS_MAX\n    for this, which makes the code a bit more generic and allows to remove\n    BPF_TUNLEN_MAX from eBPF code.\n\n    Signed-off-by: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit e019d113b5b70845796fb2743f2504e81d7dbdfe\nAuthor: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\nDate:   Wed Mar 16 01:42:50 2016 +0100\n\n    bpf, dst: add and use dst_tclassid helper\n\n    We can just add a small helper dst_tclassid() for retrieving the\n    dst-\u003etclassid value. It makes the code a bit better in that we can\n    get rid of the ifdef from filter.c by moving this into the header.\n\n    Signed-off-by: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 5444ebc244dde214add50d81d3b0727a90c09526\nAuthor: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\nDate:   Wed Mar 16 01:42:49 2016 +0100\n\n    bpf: make skb-\u003etc_classid also readable\n\n    Currently, the tc_classid from eBPF skb context is write-only, but there\u0027s\n    no good reason for tc programs to limit it to write-only. For example,\n    it can be used to transfer its state via tail calls where the resulting\n    tc_classid gets filled gradually.\n\n    Signed-off-by: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Acked-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 28c3e3f446ca221938b8467911f00ee4b8d7c227\nAuthor: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\nDate:   Wed Mar 9 03:00:05 2016 +0100\n\n    bpf: support flow label for bpf_skb_{set, get}_tunnel_key\n\n    This patch extends bpf_tunnel_key with a tunnel_label member, that maps\n    to ip_tunnel_key\u0027s label so underlying backends like vxlan and geneve\n    can propagate the label to udp_tunnel6_xmit_skb(), where it\u0027s being set\n    in the IPv6 header. It allows for having 20 more bits to encode/decode\n    flow related meta information programmatically. Tested with vxlan and\n    geneve.\n\n    Signed-off-by: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Acked-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 923f65de93924a0073f5a6a5c62330e9f1ea1f2a\nAuthor: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\nDate:   Fri Mar 4 15:15:06 2016 +0100\n\n    bpf: support for access to tunnel options\n\n    After eBPF being able to programmatically access/manage tunnel key meta\n    data via commit d3aa45ce6b94 (\"bpf: add helpers to access tunnel metadata\")\n    and more recently also for IPv6 through c6c33454072f (\"bpf: support ipv6\n    for bpf_skb_{set,get}_tunnel_key\"), this work adds two complementary\n    helpers to generically access their auxiliary tunnel options.\n\n    Geneve and vxlan support this facility. For geneve, TLVs can be pushed,\n    and for the vxlan case its GBP extension. I.e. setting tunnel key for geneve\n    case only makes sense, if we can also read/write TLVs into it. In the GBP\n    case, it provides the flexibility to easily map the group policy ID in\n    combination with other helpers or maps.\n\n    I chose to model this as two separate helpers, bpf_skb_{set,get}_tunnel_opt(),\n    for a couple of reasons. bpf_skb_{set,get}_tunnel_key() is already rather\n    complex by itself, and there may be cases for tunnel key backends where\n    tunnel options are not always needed. If we would have integrated this\n    into bpf_skb_{set,get}_tunnel_key() nevertheless, we are very limited with\n    remaining helper arguments, so keeping compatibility on structs in case of\n    passing in a flat buffer gets more cumbersome. Separating both also allows\n    for more flexibility and future extensibility, f.e. options could be fed\n    directly from a map, etc.\n\n    Moreover, change geneve\u0027s xmit path to test only for info-\u003eoptions_len\n    instead of TUNNEL_GENEVE_OPT flag. This makes it more consistent with vxlan\u0027s\n    xmit path and allows for avoiding to specify a protocol flag in the API on\n    xmit, so it can be protocol agnostic. Having info-\u003eoptions_len is enough\n    information that is needed. Tested with vxlan and geneve.\n\n    Signed-off-by: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Acked-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit b84ab931f1eb8bdfe3e6d49329fa81fec8f98127\nAuthor: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\nDate:   Fri Mar 4 15:15:05 2016 +0100\n\n    bpf: allow to propagate df in bpf_skb_set_tunnel_key\n\n    Added by 9a628224a61b (\"ip_tunnel: Add dont fragment flag.\"), allow to\n    feed df flag into tunneling facilities (currently supported on TX by\n    vxlan, geneve and gre) as a hint from eBPF\u0027s bpf_skb_set_tunnel_key()\n    helper.\n\n    Signed-off-by: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Acked-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit eedb767645d93b91d544ba6e0f527f331115fb6a\nAuthor: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\nDate:   Fri Mar 4 15:15:04 2016 +0100\n\n    bpf: make helper function protos static\n\n    They are only used here, so there\u0027s no reason they should not be static.\n    Only the vlan push/pop protos are used in the test_bpf suite.\n\n    Signed-off-by: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Acked-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 77cb4bc19132828731d7a26a5940c0e17f88f560\nAuthor: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\nDate:   Fri Mar 4 15:15:03 2016 +0100\n\n    bpf: add flags to bpf_skb_store_bytes for clearing hash\n\n    When overwriting parts of the packet with bpf_skb_store_bytes() that\n    were fed previously into skb-\u003ehash calculation, we should clear the\n    current hash with skb_clear_hash(), so that a next skb_get_hash() call\n    can determine the correct hash related to this skb.\n\n    Signed-off-by: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Acked-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 974c432db5897310d88c9b2d13e925b4f1449ef6\nAuthor: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\nDate:   Fri Mar 4 15:15:02 2016 +0100\n\n    bpf: allow bpf_csum_diff to feed bpf_l3_csum_replace as well\n\n    Commit 7d672345ed29 (\"bpf: add generic bpf_csum_diff helper\") added a\n    generic checksum diff helper that can feed bpf_l4_csum_replace() with\n    a target __wsum diff that is to be applied to the L4 checksum. This\n    facility is very flexible, can be cascaded, allows for adding, removing,\n    or diffing data, or for calculating the pseudo header checksum from\n    scratch, but it can also be reused for working with the IPv4 header\n    checksum.\n\n    Thus, analogous to bpf_l4_csum_replace(), add a case for header field\n    value of 0 to change the checksum at a given offset through a new helper\n    csum_replace_by_diff(). Also, in addition to that, this provides an\n    easy to use interface for feeding precalculated diffs f.e. coming from\n    a map. It nicely complements bpf_l3_csum_replace() that currently allows\n    only for csum updates of 2 and 4 byte diffs.\n\n    Signed-off-by: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Acked-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 75ab2623952b8a8a7daf687f49dd75ecf7bd01d0\nAuthor: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\nDate:   Tue Feb 23 02:05:26 2016 +0100\n\n    bpf: fix csum setting for bpf_set_tunnel_key\n\n    The fix in 35e2d1152b22 (\"tunnels: Allow IPv6 UDP checksums to be correctly\n    controlled.\") changed behavior for bpf_set_tunnel_key() when in use with\n    IPv6 and thus uncovered a bug that TUNNEL_CSUM needed to be set but wasn\u0027t.\n    As a result, the stack dropped ingress vxlan IPv6 packets, that have been\n    sent via eBPF through collect meta data mode due to checksum now being zero.\n\n    Since after LCO, we enable IPv4 checksum by default, so make that analogous\n    and only provide a flag BPF_F_ZERO_CSUM_TX for the user to turn it off in\n    IPv4 case.\n\n    Fixes: 35e2d1152b22 (\"tunnels: Allow IPv6 UDP checksums to be correctly controlled.\")\n    Fixes: c6c33454072f (\"bpf: support ipv6 for bpf_skb_{set,get}_tunnel_key\")\n    Signed-off-by: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Acked-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 2d51285ea06be0bca4a25fd7f59cbae147e5fbd2\nAuthor: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\nDate:   Fri Feb 19 23:05:26 2016 +0100\n\n    bpf: fix csum update in bpf_l4_csum_replace helper for udp\n\n    When using this helper for updating UDP checksums, we need to extend\n    this in order to write CSUM_MANGLED_0 for csum computations that result\n    into 0 as sum. Reason we need this is because packets with a checksum\n    could otherwise become incorrectly marked as a packet without a checksum.\n    Likewise, if the user indicates BPF_F_MARK_MANGLED_0, then we should\n    not turn packets without a checksum into ones with a checksum.\n\n    Signed-off-by: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Acked-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit b2b2f39d94bb8b2e1ff5011a0ed41aeb69ce4ef5\nAuthor: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\nDate:   Fri Feb 19 23:05:27 2016 +0100\n\n    bpf: don\u0027t emit mov A,A on return\n\n    While debugging with bpf_jit_disasm I noticed emissions of \u0027mov %eax,%eax\u0027,\n    and found that this comes from BPF_RET | BPF_A translations from classic\n    BPF. Emitting this is unnecessary as BPF_REG_A is mapped into BPF_REG_0\n    already, therefore only emit a mov when immediates are used as return value.\n\n    Signed-off-by: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Acked-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 0eb97f5f8382536f8f7dbe4fe805a9c45252b01f\nAuthor: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\nDate:   Fri Feb 19 23:05:24 2016 +0100\n\n    bpf: remove artificial bpf_skb_{load, store}_bytes buffer limitation\n\n    We currently limit bpf_skb_store_bytes() and bpf_skb_load_bytes()\n    helpers to only store or load a maximum buffer of 16 bytes. Thus,\n    loading, rewriting and storing headers require several bpf_skb_load_bytes()\n    and bpf_skb_store_bytes() calls.\n\n    Also here we can use a per-cpu scratch buffer instead in order to not\n    pressure stack space any further. I do suspect that this limit was mainly\n    set in place for this particular reason. So, ease program development\n    by removing this limitation and make the scratchpad generic, so it can\n    be reused.\n\n    Signed-off-by: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Acked-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit d6ccffed315d984af89f1ff4c41e55b2d1123c69\nAuthor: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\nDate:   Fri Feb 19 23:05:23 2016 +0100\n\n    bpf: add generic bpf_csum_diff helper\n\n    For L4 checksums, we currently have bpf_l4_csum_replace() helper. It\u0027s\n    currently limited to handle 2 and 4 byte changes in a header and feeds the\n    from/to into inet_proto_csum_replace{2,4}() helpers of the kernel. When\n    working with IPv6, for example, this makes it rather cumbersome to deal\n    with, similarly when editing larger parts of a header.\n\n    Instead, extend the API in a more generic way: For bpf_l4_csum_replace(),\n    add a case for header field mask of 0 to change the checksum at a given\n    offset through inet_proto_csum_replace_by_diff(), and provide a helper\n    bpf_csum_diff() that can generically calculate a from/to diff for arbitrary\n    amounts of data.\n\n    This can be used in multiple ways: for the bpf_l4_csum_replace() only\n    part, this even provides us with the option to insert precalculated diffs\n    from user space f.e. from a map, or from bpf_csum_diff() during runtime.\n\n    bpf_csum_diff() has a optional from/to stack buffer input, so we can\n    calculate a diff by using a scratchbuffer for scenarios where we\u0027re\n    inserting (from is NULL), removing (to is NULL) or diffing (from/to buffers\n    don\u0027t need to be of equal size) data. Also, bpf_csum_diff() allows to\n    feed a previous csum into csum_partial(), so the function can also be\n    cascaded.\n\n    Signed-off-by: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Acked-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit fc4e2876b85bccc61112a411a4e0e9ca5c2c69e3\nAuthor: Hannes Frederic Sowa \u003channes@stressinduktion.org\u003e\nDate:   Tue Apr 5 17:10:16 2016 +0200\n\n    tun: use socket locks for sk_{attach,detatch}_filter\n\n    This reverts commit 5a5abb1fa3b05dd (\"tun, bpf: fix suspicious RCU usage\n    in tun_{attach, detach}_filter\") and replaces it to use lock_sock around\n    sk_{attach,detach}_filter. The checks inside filter.c are updated with\n    lockdep_sock_is_held to check for proper socket locks.\n\n    It keeps the code cleaner by ensuring that only one lock governs the\n    socket filter instead of two independent locks.\n\n    Cc: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Signed-off-by: Hannes Frederic Sowa \u003channes@stressinduktion.org\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit b43281711b782ed05f3527b935f5f532889561e7\nAuthor: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\nDate:   Wed Apr 13 00:10:51 2016 +0200\n\n    bpf, verifier: add ARG_PTR_TO_RAW_STACK type\n\n    When passing buffers from eBPF stack space into a helper function, we have\n    ARG_PTR_TO_STACK argument type for helpers available. The verifier makes sure\n    that such buffers are initialized, within boundaries, etc.\n\n    However, the downside with this is that we have a couple of helper functions\n    such as bpf_skb_load_bytes() that fill out the passed buffer in the expected\n    success case anyway, so zero initializing them prior to the helper call is\n    unneeded/wasted instructions in the eBPF program that can be avoided.\n\n    Therefore, add a new helper function argument type called ARG_PTR_TO_RAW_STACK.\n    The idea is to skip the STACK_MISC check in check_stack_boundary() and color\n    the related stack slots as STACK_MISC after we checked all call arguments.\n\n    Helper functions using ARG_PTR_TO_RAW_STACK must make sure that every path of\n    the helper function will fill the provided buffer area, so that we cannot leak\n    any uninitialized stack memory. This f.e. means that error paths need to\n    memset() the buffers, but the expected fast-path doesn\u0027t have to do this\n    anymore.\n\n    Since there\u0027s no such helper needing more than at most one ARG_PTR_TO_RAW_STACK\n    argument, we can keep it simple and don\u0027t need to check for multiple areas.\n    Should in future such a use-case really appear, we have check_raw_mode() that\n    will make sure we implement support for it first.\n\n    Signed-off-by: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Acked-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit df1a5c0530d53fa65b8ab5b2d24d28b3a92455b3\nAuthor: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\nDate:   Wed Apr 13 00:10:50 2016 +0200\n\n    bpf, verifier: add bpf_call_arg_meta for passing meta data\n\n    Currently, when the verifier checks calls in check_call() function, we\n    call check_func_arg() for all 5 arguments e.g. to make sure expected types\n    are correct. In some cases, we collect meta data (here: map pointer) to\n    perform additional checks such as checking stack boundary on key/value\n    sizes for subsequent arguments. As we\u0027re going to extend the meta data,\n    add a generic struct bpf_call_arg_meta that we can use for passing into\n    check_func_arg().\n\n    Signed-off-by: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Acked-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit d87840f6beec118e1bfd78ed787218477b8c30d9\nAuthor: Alexei Starovoitov \u003cast@fb.com\u003e\nDate:   Tue Apr 12 10:26:19 2016 -0700\n\n    bpf/verifier: reject invalid LD_ABS | BPF_DW instruction\n\n    verifier must check for reserved size bits in instruction opcode and\n    reject BPF_LD | BPF_ABS | BPF_DW and BPF_LD | BPF_IND | BPF_DW instructions,\n    otherwise interpreter will WARN_RATELIMIT on them during execution.\n\n    Fixes: ddd872bc3098 (\"bpf: verifier: add checks for BPF_ABS | BPF_IND instructions\")\n    Signed-off-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Acked-by: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 9d4e36119e9e9a802f6687bb12ae302526bebac9\nAuthor: Alexei Starovoitov \u003cast@fb.com\u003e\nDate:   Wed Apr 6 19:39:21 2016 -0700\n\n    bpf: simplify verifier register state assignments\n\n    verifier is using the following structure to track the state of registers:\n    struct reg_state {\n        enum bpf_reg_type type;\n        union {\n            int imm;\n            struct bpf_map *map_ptr;\n        };\n    };\n    and later on in states_equal() does memcmp(\u0026old-\u003eregs[i], \u0026cur-\u003eregs[i],..)\n    to find equivalent states.\n    Throughout the code of verifier there are assignements to \u0027imm\u0027 and \u0027map_ptr\u0027\n    fields and it\u0027s not obvious that most of the assignments into \u0027imm\u0027 don\u0027t\n    need to clear extra 4 bytes (like mark_reg_unknown_value() does) to make sure\n    that memcmp doesn\u0027t go over junk left from \u0027map_ptr\u0027 assignment.\n\n    Simplify the code by converting \u0027int\u0027 into \u0027long\u0027\n\n    Suggested-by: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Signed-off-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Acked-by: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit bed5bdc5017015346e6f9dbc47f4af6029bc1c66\nAuthor: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\nDate:   Tue Apr 5 22:33:17 2016 +0200\n\n    bpf, verifier: further improve search pruning\n\n    The verifier needs to go through every path of the program in\n    order to check that it terminates safely, which can be quite a\n    lot of instructions that need to be processed f.e. in cases with\n    more branchy programs. With search pruning from f1bca824dabb (\"bpf:\n    add search pruning optimization to verifier\") the search space can\n    already be reduced significantly when the verifier detects that\n    a previously walked path with same register and stack contents\n    terminated already (see verifier\u0027s states_equal()), so the search\n    can skip walking those states.\n\n    When working with larger programs of \u003e ~2000 (out of max 4096)\n    insns, we found that the current limit of 32k instructions is easily\n    hit. For example, a case we ran into is that the search space cannot\n    be pruned due to branches at the beginning of the program that make\n    use of certain stack space slots (STACK_MISC), which are never used\n    in the remaining program (STACK_INVALID). Therefore, the verifier\n    needs to walk paths for the slots in STACK_INVALID state, but also\n    all remaining paths with a stack structure, where the slots are in\n    STACK_MISC, which can nearly double the search space needed. After\n    various experiments, we find that a limit of 64k processed insns is\n    a more reasonable choice when dealing with larger programs in practice.\n    This still allows to reject extreme crafted cases that can have a\n    much higher complexity (f.e. \u003e ~300k) within the 4096 insns limit\n    due to search pruning not being able to take effect.\n\n    Furthermore, we found that a lot of states can be pruned after a\n    call instruction, f.e. we were able to reduce the search state by\n    ~35% in some cases with this heuristic, trade-off is to keep a bit\n    more states in env-\u003eexplored_states. Usually, call instructions\n    have a number of preceding register assignments and/or stack stores,\n    where search pruning has a better chance to suceed in states_equal()\n    test. The current code marks the branch targets with STATE_LIST_MARK\n    in case of conditional jumps, and the next (t + 1) instruction in\n    case of unconditional jump so that f.e. a backjump will walk it. We\n    also did experiments with using t + insns[t].off + 1 as a marker in\n    the unconditionally jump case instead of t + 1 with the rationale\n    that these two branches of execution that converge after the label\n    might have more potential of pruning. We found that it was a bit\n    better, but not necessarily significantly better than the current\n    state, perhaps also due to clang not generating back jumps often.\n    Hence, we left that as is for now.\n\n    Signed-off-by: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Acked-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 52f14bd28217ac0adf4f4f6361328b8169d84985\nAuthor: Alexei Starovoitov \u003cast@fb.com\u003e\nDate:   Wed Apr 6 18:43:28 2016 -0700\n\n    bpf: sanitize bpf tracepoint access\n\n    during bpf program loading remember the last byte of ctx access\n    and at the time of attaching the program to tracepoint check that\n    the program doesn\u0027t access bytes beyond defined in tracepoint fields\n\n    This also disallows access to __dynamic_array fields, but can be\n    relaxed in the future.\n\n    Signed-off-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 8d9e6be996bd26c1cae1293b03f214558291161d\nAuthor: Alexei Starovoitov \u003cast@fb.com\u003e\nDate:   Wed Apr 6 18:43:27 2016 -0700\n\n    bpf: support bpf_get_stackid() and bpf_perf_event_output() in tracepoint programs\n\n    needs two wrapper functions to fetch \u0027struct pt_regs *\u0027 to convert\n    tracepoint bpf context into kprobe bpf context to reuse existing\n    helper functions\n\n    Signed-off-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 5f3daf5505858a053dd913a489708908a8f425de\nAuthor: Alexei Starovoitov \u003cast@fb.com\u003e\nDate:   Wed Apr 6 18:43:26 2016 -0700\n\n    bpf: register BPF_PROG_TYPE_TRACEPOINT program type\n\n    register tracepoint bpf program type and let it call the same set\n    of helper functions as BPF_PROG_TYPE_KPROBE\n\n    Signed-off-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 690fc43c27db20255a329aa348a9656eb29fc71f\nAuthor: Alexei Starovoitov \u003cast@fb.com\u003e\nDate:   Wed Apr 6 18:43:25 2016 -0700\n\n    perf, bpf: allow bpf programs attach to tracepoints\n\n    introduce BPF_PROG_TYPE_TRACEPOINT program type and allow it to be attached\n    to the perf tracepoint handler, which will copy the arguments into\n    the per-cpu buffer and pass it to the bpf program as its first argument.\n    The layout of the fields can be discovered by doing\n    \u0027cat /sys/kernel/debug/tracing/events/sched/sched_switch/format\u0027\n    prior to the compilation of the program with exception that first 8 bytes\n    are reserved and not accessible to the program. This area is used to store\n    the pointer to \u0027struct pt_regs\u0027 which some of the bpf helpers will use:\n    +---------+\n    | 8 bytes | hidden \u0027struct pt_regs *\u0027 (inaccessible to bpf program)\n    +---------+\n    | N bytes | static tracepoint fields defined in tracepoint/format (bpf readonly)\n    +---------+\n    | dynamic | __dynamic_array bytes of tracepoint (inaccessible to bpf yet)\n    +---------+\n\n    Not that all of the fields are already dumped to user space via perf ring buffer\n    and broken application access it directly without consulting tracepoint/format.\n    Same rule applies here: static tracepoint fields should only be accessed\n    in a format defined in tracepoint/format. The order of fields and\n    field sizes are not an ABI.\n\n    Signed-off-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Acked-by: Peter Zijlstra (Intel) \u003cpeterz@infradead.org\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit f843bdb1d837e517dfa79a0cd049266991f60805\nAuthor: Alexei Starovoitov \u003cast@fb.com\u003e\nDate:   Wed Apr 6 18:43:24 2016 -0700\n\n    perf: split perf_trace_buf_prepare into alloc and update parts\n\n    split allows to move expensive update of \u0027struct trace_entry\u0027 to later phase.\n    Repurpose unused 1st argument of perf_tp_event() to indicate event type.\n\n    While splitting use temp variable \u0027rctx\u0027 instead of \u0027*rctx\u0027 to avoid\n    unnecessary loads done by the compiler due to -fno-strict-aliasing\n\n    Signed-off-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Acked-by: Peter Zijlstra (Intel) \u003cpeterz@infradead.org\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit bcc6d4666142eb01096996dea52de927b30738f0\nAuthor: Alexei Starovoitov \u003cast@fb.com\u003e\nDate:   Wed Apr 6 18:43:23 2016 -0700\n\n    perf: remove unused __addr variable\n\n    now all calls to perf_trace_buf_submit() pass 0 as 4th\n    argument which will be repurposed in the next patch which will\n    change the meaning of 1st arg of perf_tp_event() to event_type\n\n    Signed-off-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Acked-by: Peter Zijlstra (Intel) \u003cpeterz@infradead.org\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit ca7310e4cabb182e3a34948574d3909795f6000b\nAuthor: Al Viro \u003cviro@zeniv.linux.org.uk\u003e\nDate:   Fri Mar 25 12:06:51 2016 -0400\n\n    bpf: reject invalid names right in -\u003elookup()\n\n    ... and other methods won\u0027t see them at all\n\n    Signed-off-by: Al Viro \u003cviro@zeniv.linux.org.uk\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 7275e7feab31b599e377fecd5b56a66abf5d6507\nAuthor: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\nDate:   Fri Mar 25 00:30:25 2016 +0100\n\n    bpf: add missing map_flags to bpf_map_show_fdinfo\n\n    Add map_flags attribute to bpf_map_show_fdinfo(), so that tools like\n    tc can check for them when loading objects from a pinned entry, e.g.\n    if user intent wrt allocation (BPF_F_NO_PREALLOC) is different to the\n    pinned object, it can bail out. Follow-up to 6c9059817432 (\"bpf:\n    pre-allocate hash map elements\"), so that tc can still support this\n    with v4.6.\n\n    Signed-off-by: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Acked-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 82c9e5299fcd5fc0af0a49e8478cc3af6f9f9488\nAuthor: Alexei Starovoitov \u003cast@fb.com\u003e\nDate:   Wed Mar 9 20:02:33 2016 -0800\n\n    bpf: avoid copying junk bytes in bpf_get_current_comm()\n\n    Lots of places in the kernel use memcpy(buf, comm, TASK_COMM_LEN); but\n    the result is typically passed to print(\"%s\", buf) and extra bytes\n    after zero don\u0027t cause any harm.\n    In bpf the result of bpf_get_current_comm() is used as the part of\n    map key and was causing spurious hash map mismatches.\n    Use strlcpy() to guarantee zero-terminated string.\n    bpf verifier checks that output buffer is zero-initialized,\n    so even for short task names the output buffer don\u0027t have junk bytes.\n    Note it\u0027s not a security concern, since kprobe+bpf is root only.\n\n    Fixes: ffeedafbf023 (\"bpf: introduce current-\u003epid, tgid, uid, gid, comm accessors\")\n    Reported-by: Tobias Waldekranz \u003ctobias@waldekranz.com\u003e\n    Signed-off-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 5697a5bbb767c3122c70833ddf5932d5f9aa88f6\nAuthor: Alexei Starovoitov \u003cast@fb.com\u003e\nDate:   Wed Mar 9 18:56:49 2016 -0800\n\n    bpf: bpf_stackmap_copy depends on CONFIG_PERF_EVENTS\n\n    0-day bot reported build error:\n    kernel/built-in.o: In function `map_lookup_elem\u0027:\n    \u003e\u003e kernel/bpf/.tmp_syscall.o:(.text+0x329b3c): undefined reference to `bpf_stackmap_copy\u0027\n    when CONFIG_BPF_SYSCALL is set and CONFIG_PERF_EVENTS is not.\n    Add weak definition to resolve it.\n    This code path in map_lookup_elem() is never taken\n    when CONFIG_PERF_EVENTS is not set.\n\n    Fixes: 557c0c6e7df8 (\"bpf: convert stackmap to pre-allocation\")\n    Reported-by: Fengguang Wu \u003cfengguang.wu@intel.com\u003e\n    Signed-off-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 1026ed0a6926d95d3df2a71b90a6ffa6a390674e\nAuthor: Alexei Starovoitov \u003cast@fb.com\u003e\nDate:   Mon Mar 7 21:57:17 2016 -0800\n\n    bpf: convert stackmap to pre-allocation\n\n    It was observed that calling bpf_get_stackid() from a kprobe inside\n    slub or from spin_unlock causes similar deadlock as with hashmap,\n    therefore convert stackmap to use pre-allocated memory.\n\n    The call_rcu is no longer feasible mechanism, since delayed freeing\n    causes bpf_get_stackid() to fail unpredictably when number of actual\n    stacks is significantly less than user requested max_entries.\n    Since elements are no longer freed into slub, we can push elements into\n    freelist immediately and let them be recycled.\n    However the very unlikley race between user space map_lookup() and\n    program-side recycling is possible:\n         cpu0                          cpu1\n         ----                          ----\n    user does lookup(stackidX)\n    starts copying ips into buffer\n                                       delete(stackidX)\n                                       calls bpf_get_stackid()\n    \t\t\t\t   which recyles the element and\n                                       overwrites with new stack trace\n\n    To avoid user space seeing a partial stack trace consisting of two\n    merged stack traces, do bucket \u003d xchg(, NULL); copy; xchg(,bucket);\n    to preserve consistent stack trace delivery to user space.\n    Now we can move memset(,0) of left-over element value from critical\n    path of bpf_get_stackid() into slow-path of user space lookup.\n    Also disallow lookup() from bpf program, since it\u0027s useless and\n    program shouldn\u0027t be messing with collected stack trace.\n\n    Note that similar race between user space lookup and kernel side updates\n    is also present in hashmap, but it\u0027s not a new race. bpf programs were\n    always allowed to modify hash and array map elements while user space\n    is copying them.\n\n    Fixes: d5a3b1f69186 (\"bpf: introduce BPF_MAP_TYPE_STACK_TRACE\")\n    Signed-off-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 4ae958ac1ffac94933ab5a3b99d21e6b90adbb99\nAuthor: Alexei Starovoitov \u003cast@fb.com\u003e\nDate:   Mon Mar 7 21:57:16 2016 -0800\n\n    bpf: check for reserved flag bits in array and stack maps\n\n    Suggested-by: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Signed-off-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit ea8a7a71bb7ddf960cb9b4c99211601aba44fcb5\nAuthor: Alexei Starovoitov \u003cast@fb.com\u003e\nDate:   Mon Mar 7 21:57:15 2016 -0800\n\n    bpf: pre-allocate hash map elements\n\n    If kprobe is placed on spin_unlock then calling kmalloc/kfree from\n    bpf programs is not safe, since the following dead lock is possible:\n    kfree-\u003espin_lock(kmem_cache_node-\u003elock)...spin_unlock-\u003ekprobe-\u003e\n    bpf_prog-\u003emap_update-\u003ekmalloc-\u003espin_lock(of the same kmem_cache_node-\u003elock)\n    and deadlocks.\n\n    The following solutions were considered and some implemented, but\n    eventually discarded\n    - kmem_cache_create for every map\n    - add recursion check to slow-path of slub\n    - use reserved memory in bpf_map_update for in_irq or in preempt_disabled\n    - kmalloc via irq_work\n\n    At the end pre-allocation of all map elements turned out to be the simplest\n    solution and since the user is charged upfront for all the memory, such\n    pre-allocation doesn\u0027t affect the user space visible behavior.\n\n    Since it\u0027s impossible to tell whether kprobe is triggered in a safe\n    location from kmalloc point of view, use pre-allocation by default\n    and introduce new BPF_F_NO_PREALLOC flag.\n\n    While testing of per-cpu hash maps it was discovered\n    that alloc_percpu(GFP_ATOMIC) has odd corner cases and often\n    fails to allocate memory even when 90% of it is free.\n    The pre-allocation of per-cpu hash elements solves this problem as well.\n\n    Turned out that bpf_map_update() quickly followed by\n    bpf_map_lookup()+bpf_map_delete() is very common pattern used\n    in many of iovisor/bcc/tools, so there is additional benefit of\n    pre-allocation, since such use cases are must faster.\n\n    Since all hash map elements are now pre-allocated we can remove\n    atomic increment of htab-\u003ecount and save few more cycles.\n\n    Also add bpf_map_precharge_memlock() to check rlimit_memlock early to avoid\n    large malloc/free done by users who don\u0027t have sufficient limits.\n\n    Pre-allocation is done with vmalloc and alloc/free is done\n    via percpu_freelist. Here are performance numbers for different\n    pre-allocation algorithms that were implemented, but discarded\n    in favor of percpu_freelist:\n\n    1 cpu:\n    pcpu_ida\t2.1M\n    pcpu_ida nolock\t2.3M\n    bt\t\t2.4M\n    kmalloc\t\t1.8M\n    hlist+spinlock\t2.3M\n    pcpu_freelist\t2.6M\n\n    4 cpu:\n    pcpu_ida\t1.5M\n    pcpu_ida nolock\t1.8M\n    bt w/smp_align\t1.7M\n    bt no/smp_align\t1.1M\n    kmalloc\t\t0.7M\n    hlist+spinlock\t0.2M\n    pcpu_freelist\t2.0M\n\n    8 cpu:\n    pcpu_ida\t0.7M\n    bt w/smp_align\t0.8M\n    kmalloc\t\t0.4M\n    pcpu_freelist\t1.5M\n\n    32 cpu:\n    kmalloc\t\t0.13M\n    pcpu_freelist\t0.49M\n\n    pcpu_ida nolock is a modified percpu_ida algorithm without\n    percpu_ida_cpu locks and without cross-cpu tag stealing.\n    It\u0027s faster than existing percpu_ida, but not as fast as pcpu_freelist.\n\n    bt is a variant of block/blk-mq-tag.c simlified and customized\n    for bpf use case. bt w/smp_align is using cache line for every \u0027long\u0027\n    (similar to blk-mq-tag). bt no/smp_align allocates \u0027long\u0027\n    bitmasks continuously to save memory. It\u0027s comparable to percpu_ida\n    and in some cases faster, but slower than percpu_freelist\n\n    hlist+spinlock is the simplest free list with single spinlock.\n    As expeceted it has very bad scaling in SMP.\n\n    kmalloc is existing implementation which is still available via\n    BPF_F_NO_PREALLOC flag. It\u0027s significantly slower in single cpu and\n    in 8 cpu setup it\u0027s 3 times slower than pre-allocation with pcpu_freelist,\n    but saves memory, so in cases where map-\u003emax_entries can be large\n    and number of map update/delete per second is low, it may make\n    sense to use it.\n\n    Signed-off-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit c50a12f07c71db6b40b81e06c8ae4327aaf7bcfb\nAuthor: Alexei Starovoitov \u003cast@fb.com\u003e\nDate:   Mon Mar 7 21:57:14 2016 -0800\n\n    bpf: introduce percpu_freelist\n\n    Introduce simple percpu_freelist to keep single list of elements\n    spread across per-cpu singly linked lists.\n\n    /* push element into the list */\n    void pcpu_freelist_push(struct pcpu_freelist *, struct pcpu_freelist_node *);\n\n    /* pop element from the list */\n    struct pcpu_freelist_node *pcpu_freelist_pop(struct pcpu_freelist *);\n\n    The object is pushed to the current cpu list.\n    Pop first trying to get the object from the current cpu list,\n    if it\u0027s empty goes to the neigbour cpu list.\n\n    For bpf program usage pattern the collision rate is very low,\n    since programs push and pop the objects typically on the same cpu.\n\n    Signed-off-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 6df06831fd6ff9e250b81f108dc3fd0ef03f6a3a\nAuthor: Alexei Starovoitov \u003cast@fb.com\u003e\nDate:   Mon Mar 7 21:57:13 2016 -0800\n\n    bpf: prevent kprobe+bpf deadlocks\n\n    if kprobe is placed within update or delete hash map helpers\n    that hold bucket spin lock and triggered bpf program is trying to\n    grab the spinlock for the same bucket on the same cpu, it will\n    deadlock.\n    Fix it by extending existing recursion prevention mechanism.\n\n    Note, map_lookup and other tracing helpers don\u0027t have this problem,\n    since they don\u0027t hold any locks and don\u0027t modify global data.\n    bpf_trace_printk has its own recursive check and ok as well.\n\n    Signed-off-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Acked-by: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit a7a635cb86b7d53895f15393091155da3d0d522e\nAuthor: Josh Poimboeuf \u003cjpoimboe@redhat.com\u003e\nDate:   Sun Feb 28 22:22:37 2016 -0600\n\n    bpf: Mark __bpf_prog_run() stack frame as non-standard\n\n    objtool reports the following false positive warnings:\n\n      kernel/bpf/core.o: warning: objtool: __bpf_prog_run()+0x5c: sibling call from callable instruction with changed frame pointer\n      kernel/bpf/core.o: warning: objtool: __bpf_prog_run()+0x60: function has unreachable instruction\n      kernel/bpf/core.o: warning: objtool: __bpf_prog_run()+0x64: function has unreachable instruction\n      [...]\n\n    It\u0027s confused by the following dynamic jump instruction in\n    __bpf_prog_run()::\n\n      jmp     *(%r12,%rax,8)\n\n    which corresponds to the following line in the C code:\n\n      goto *jumptable[insn-\u003ecode];\n\n    There\u0027s no way for objtool to deterministically find all possible\n    branch targets for a dynamic jump, so it can\u0027t verify this code.\n\n    In this case the jumps all stay within the function, and there\u0027s nothing\n    unusual going on related to the stack, so we can whitelist the function.\n\n    Signed-off-by: Josh Poimboeuf \u003cjpoimboe@redhat.com\u003e\n    Acked-by: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Acked-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Cc: Andrew Morton \u003cakpm@linux-foundation.org\u003e\n    Cc: Andy Lutomirski \u003cluto@kernel.org\u003e\n    Cc: Arnaldo Carvalho de Melo \u003cacme@kernel.org\u003e\n    Cc: Bernd Petrovitsch \u003cbernd@petrovitsch.priv.at\u003e\n    Cc: Borislav Petkov \u003cbp@alien8.de\u003e\n    Cc: Chris J Arges \u003cchris.j.arges@canonical.com\u003e\n    Cc: Jiri Slaby \u003cjslaby@suse.cz\u003e\n    Cc: Linus Torvalds \u003ctorvalds@linux-foundation.org\u003e\n    Cc: Michal Marek \u003cmmarek@suse.cz\u003e\n    Cc: Namhyung Kim \u003cnamhyung@gmail.com\u003e\n    Cc: Pedro Alves \u003cpalves@redhat.com\u003e\n    Cc: Peter Zijlstra \u003cpeterz@infradead.org\u003e\n    Cc: Thomas Gleixner \u003ctglx@linutronix.de\u003e\n    Cc: live-patching@vger.kernel.org\n    Cc: netdev@vger.kernel.org\n    Link: http://lkml.kernel.org/r/b90e6bf3fdbfb5c4cc1b164b965502e53cf48935.1456719558.git.jpoimboe@redhat.com\n    Signed-off-by: Ingo Molnar \u003cmingo@kernel.org\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 725745db5229fe3cb48146d645713b4bcf659554\nAuthor: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\nDate:   Fri Feb 19 23:05:22 2016 +0100\n\n    bpf: add new arg_type that allows for 0 sized stack buffer\n\n    Currently, when we pass a buffer from the eBPF stack into a helper\n    function, the function proto indicates argument types as ARG_PTR_TO_STACK\n    and ARG_CONST_STACK_SIZE pair. If R\u003cX\u003e contains the former, then R\u003cX+1\u003e\n    must be of the latter type. Then, verifier checks whether the buffer\n    points into eBPF stack, is initialized, etc. The verifier also guarantees\n    that the constant value passed in R\u003cX+1\u003e is greater than 0, so helper\n    functions don\u0027t need to test for it and can always assume a non-NULL\n    initialized buffer as well as non-0 buffer size.\n\n    This patch adds a new argument types ARG_CONST_STACK_SIZE_OR_ZERO that\n    allows to also pass NULL as R\u003cX\u003e and 0 as R\u003cX+1\u003e into the helper function.\n    Such helper functions, of course, need to be able to handle these cases\n    internally then. Verifier guarantees that either R\u003cX\u003e \u003d\u003d NULL \u0026\u0026 R\u003cX+1\u003e \u003d\u003d 0\n    or R\u003cX\u003e !\u003d NULL \u0026\u0026 R\u003cX+1\u003e !\u003d 0 (like the case of ARG_CONST_STACK_SIZE), any\n    other combinations are not possible to load.\n\n    I went through various options of extending the verifier, and introducing\n    the type ARG_CONST_STACK_SIZE_OR_ZERO seems to have most minimal changes\n    needed to the verifier.\n\n    Signed-off-by: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Acked-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 0865cc49ae5d443a5e545651d790b09a773de0d8\nAuthor: Alexei Starovoitov \u003cast@fb.com\u003e\nDate:   Wed Feb 17 19:58:58 2016 -0800\n\n    bpf: introduce BPF_MAP_TYPE_STACK_TRACE\n\n    add new map type to store stack traces and corresponding helper\n    bpf_get_stackid(ctx, map, flags) - walk user or kernel stack and return id\n    @ctx: struct pt_regs*\n    @map: pointer to stack_trace map\n    @flags: bits 0-7 - numer of stack frames to skip\n            bit 8 - collect user stack instead of kernel\n            bit 9 - compare stacks by hash only\n            bit 10 - if two different stacks hash into the same stackid\n                     discard old\n            other bits - reserved\n    Return: \u003e\u003d 0 stackid on success or negative error\n\n    stackid is a 32-bit integer handle that can be further combined with\n    other data (including other stackid) and used as a key into maps.\n\n    Userspace will access stackmap using standard lookup/delete syscall commands to\n    retrieve full stack trace for given stackid.\n\n    Signed-off-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 6b2ced7666965407fd3f6b871fe805f40ad1294a\nAuthor: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\nDate:   Mon Jan 11 01:16:39 2016 +0100\n\n    bpf: support ipv6 for bpf_skb_{set,get}_tunnel_key\n\n    After IPv6 support has recently been added to metadata dst and related\n    encaps, add support for populating/reading it from an eBPF program.\n\n    Commit d3aa45ce6b (\"bpf: add helpers to access tunnel metadata\") started\n    with initial IPv4-only support back then (due to IPv6 metadata support\n    not being available yet).\n\n    To stay compatible with older programs, we need to test for the passed\n    structure size. Also TOS and TTL support from the ip_tunnel_info key has\n    been added. Tested with vxlan devs in collect meta data mode with IPv4,\n    IPv6 and in compat mode over different network namespaces.\n\n    Signed-off-by: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Acked-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit b3acb130defb2fa2cf9c70c659889546eba3aa64\nAuthor: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\nDate:   Mon Jan 11 01:16:38 2016 +0100\n\n    bpf: export helper function flags and reject invalid ones\n\n    Export flags used by eBPF helper functions through UAPI, so they can be\n    used by programs (instead of them redefining all flags each time or just\n    using the hard-coded values). It also gives a better overview what flags\n    are used where and we can further get rid of the extra macros defined in\n    filter.c. Moreover, reject invalid flags.\n\n    Signed-off-by: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Acked-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 4d5b01da36d112cee5943e677276d509cca68b6f\nAuthor: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\nDate:   Thu Jan 7 15:50:23 2016 +0100\n\n    bpf: add skb_postpush_rcsum and fix dev_forward_skb occasions\n\n    Add a small helper skb_postpush_rcsum() and fix up redirect locations\n    that need CHECKSUM_COMPLETE fixups on ingress. dev_forward_skb() expects\n    a proper csum that covers also Ethernet header, f.e. since 2c26d34bbcc0\n    (\"net/core: Handle csum for CHECKSUM_COMPLETE VXLAN forwarding\"), we\n    also do skb_postpull_rcsum() after pulling Ethernet header off via\n    eth_type_trans().\n\n    When using eBPF in a netns setup f.e. with vxlan in collect metadata mode,\n    I can trigger the following csum issue with an IPv6 setup:\n\n      [  505.144065] dummy1: hw csum failure\n      [...]\n      [  505.144108] Call Trace:\n      [  505.144112]  \u003cIRQ\u003e  [\u003cffffffff81372f08\u003e] dump_stack+0x44/0x5c\n      [  505.144134]  [\u003cffffffff81607cea\u003e] netdev_rx_csum_fault+0x3a/0x40\n      [  505.144142]  [\u003cffffffff815fee3f\u003e] __skb_checksum_complete+0xcf/0xe0\n      [  505.144149]  [\u003cffffffff816f0902\u003e] nf_ip6_checksum+0xb2/0x120\n      [  505.144161]  [\u003cffffffffa08c0e0e\u003e] icmpv6_error+0x17e/0x328 [nf_conntrack_ipv6]\n      [  505.144170]  [\u003cffffffffa0898eca\u003e] ? ip6t_do_table+0x2fa/0x645 [ip6_tables]\n      [  505.144177]  [\u003cffffffffa08c0725\u003e] ? ipv6_get_l4proto+0x65/0xd0 [nf_conntrack_ipv6]\n      [  505.144189]  [\u003cffffffffa06c9a12\u003e] nf_conntrack_in+0xc2/0x5a0 [nf_conntrack]\n      [  505.144196]  [\u003cffffffffa08c039c\u003e] ipv6_conntrack_in+0x1c/0x20 [nf_conntrack_ipv6]\n      [  505.144204]  [\u003cffffffff8164385d\u003e] nf_iterate+0x5d/0x70\n      [  505.144210]  [\u003cffffffff816438d6\u003e] nf_hook_slow+0x66/0xc0\n      [  505.144218]  [\u003cffffffff816bd302\u003e] ipv6_rcv+0x3f2/0x4f0\n      [  505.144225]  [\u003cffffffff816bca40\u003e] ? ip6_make_skb+0x1b0/0x1b0\n      [  505.144232]  [\u003cffffffff8160b77b\u003e] __netif_receive_skb_core+0x36b/0x9a0\n      [  505.144239]  [\u003cffffffff8160bdc8\u003e] ? __netif_receive_skb+0x18/0x60\n      [  505.144245]  [\u003cffffffff8160bdc8\u003e] __netif_receive_skb+0x18/0x60\n      [  505.144252]  [\u003cffffffff8160ccff\u003e] process_backlog+0x9f/0x140\n      [  505.144259]  [\u003cffffffff8160c4a5\u003e] net_rx_action+0x145/0x320\n      [...]\n\n    What happens is that on ingress, we push Ethernet header back in, either\n    from cls_bpf or right before skb_do_redirect(), but without updating csum.\n    The \"hw csum failure\" can be fixed by using the new skb_postpush_rcsum()\n    helper for the dev_forward_skb() case to correct the csum diff again.\n\n    Thanks to Hannes Frederic Sowa for the csum_partial() idea!\n\n    Fixes: 3896d655f4d4 (\"bpf: introduce bpf_clone_redirect() helper\")\n    Fixes: 27b29f63058d (\"bpf: add bpf_redirect() helper\")\n    Signed-off-by: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Acked-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit b130cee50164c7b2569ef4ba30c2fda072575266\nAuthor: Craig Gallek \u003ckraig@google.com\u003e\nDate:   Mon Jan 4 17:41:47 2016 -0500\n\n    soreuseport: setsockopt SO_ATTACH_REUSEPORT_[CE]BPF\n\n    Expose socket options for setting a classic or extended BPF program\n    for use when selecting sockets in an SO_REUSEPORT group.  These options\n    can be used on the first socket to belong to a group before bind or\n    on any socket in the group after bind.\n\n    This change includes refactoring of the existing sk_filter code to\n    allow reuse of the existing BPF filter validation checks.\n\n    Signed-off-by: Craig Gallek \u003ckraig@google.com\u003e\n    Acked-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n    Change-Id: If4bd3258a57eeca4622a135d9cdf64ecc12f8a27\n\ncommit de04479341b0e157354acec95983419ef9949422\nAuthor: Craig Gallek \u003ckraig@google.com\u003e\nDate:   Mon Jan 4 17:41:46 2016 -0500\n\n    soreuseport: fast reuseport UDP socket selection\n\n    Include a struct sock_reuseport instance when a UDP socket binds to\n    a specific address for the first time with the reuseport flag set.\n    When selecting a socket for an incoming UDP packet, use the information\n    available in sock_reuseport if present.\n\n    This required adding an additional field to the UDP source address\n    equality function to differentiate between exact and wildcard matches.\n    The original use case allowed wildcard matches when checking for\n    existing port uses during bind.  The new use case of adding a socket\n    to a reuseport group requires exact address matching.\n\n    Performance test (using a machine with 2 CPU sockets and a total of\n    48 cores):  Create reuseport groups of varying size.  Use one socket\n    from this group per user thread (pinning each thread to a different\n    core) calling recvmmsg in a tight loop.  Record number of messages\n    received per second while saturating a 10G link.\n      10 sockets: 18% increase (~2.8M -\u003e 3.3M pkts/s)\n      20 sockets: 14% increase (~2.9M -\u003e 3.3M pkts/s)\n      40 sockets: 13% increase (~3.0M -\u003e 3.4M pkts/s)\n\n    This work is based off a similar implementation written by\n    Ying Cai \u003cycai@google.com\u003e for implementing policy-based reuseport\n    selection.\n\n    Signed-off-by: Craig Gallek \u003ckraig@google.com\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 7a9eff8c0dc69e3fb06cd8e73b83119251b3537b\nAuthor: Craig Gallek \u003ckraig@google.com\u003e\nDate:   Mon Jan 4 17:41:45 2016 -0500\n\n    soreuseport: define reuseport groups\n\n    struct sock_reuseport is an optional shared structure referenced by each\n    socket belonging to a reuseport group.  When a socket is bound to an\n    address/port not yet in use and the reuseport flag has been set, the\n    structure will be allocated and attached to the newly bound socket.\n    When subsequent calls to bind are made for the same address/port, the\n    shared structure will be updated to include the new socket and the\n    newly bound socket will reference the group structure.\n\n    Usually, when an incoming packet was destined for a reuseport group,\n    all sockets in the same group needed to be considered before a\n    dispatching decision was made.  With this structure, an appropriate\n    socket can be found after looking up just one socket in the group.\n\n    This shared structure will also allow for more complicated decisions to\n    be made when selecting a socket (eg a BPF filter).\n\n    This work is based off a similar implementation written by\n    Ying Cai \u003cycai@google.com\u003e for implementing policy-based reuseport\n    selection.\n\n    Signed-off-by: Craig Gallek \u003ckraig@google.com\u003e\n    Acked-by: Eric Dumazet \u003cedumazet@google.com\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 48fcd84c142a36716d08e225b61fdfbb085a2d01\nAuthor: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\nDate:   Thu Dec 17 23:51:55 2015 +0100\n\n    bpf: fix misleading comment in bpf_convert_filter\n\n    Comment says \"User BPF\u0027s register A is mapped to our BPF register 6\",\n    which is actually wrong as the mapping is on register 0. This can\n    already be inferred from the code itself. So just remove it before\n    someone makes assumptions based on that. Only code tells truth. ;)\n\n    Signed-off-by: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Acked-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit f505a7c200c262fffb41fdaa7c40989a6aebe228\nAuthor: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\nDate:   Thu Dec 17 23:51:53 2015 +0100\n\n    bpf: add bpf_skb_load_bytes helper\n\n    When hacking tc programs with eBPF, one of the issues that come up\n    from time to time is to load addresses from headers. In eBPF as in\n    classic BPF, we have BPF_LD | BPF_ABS | BPF_{B,H,W} instructions that\n    extract a byte, half-word or word out of the skb data though helpers\n    such as bpf_load_pointer() (interpreter case).\n\n    F.e. extracting a whole IPv6 address could possibly look like ...\n\n      union v6addr {\n        struct {\n          __u32 p1;\n          __u32 p2;\n          __u32 p3;\n          __u32 p4;\n        };\n        __u8 addr[16];\n      };\n\n      [...]\n\n      a.p1 \u003d htonl(load_word(skb, off));\n      a.p2 \u003d htonl(load_word(skb, off +  4));\n      a.p3 \u003d htonl(load_word(skb, off +  8));\n      a.p4 \u003d htonl(load_word(skb, off + 12));\n\n      [...]\n\n      /* access to a.addr[...] */\n\n    This work adds a complementary helper bpf_skb_load_bytes() (we also\n    have bpf_skb_store_bytes()) as an alternative where the same call\n    would look like from an eBPF program:\n\n      ret \u003d bpf_skb_load_bytes(skb, off, addr, sizeof(addr));\n\n    Same verifier restrictions apply as in ffeedafbf023 (\"bpf: introduce\n    current-\u003epid, tgid, uid, gid, comm accessors\") case, where stack memory\n    access needs to be statically verified and thus guaranteed to be\n    initialized in first use (otherwise verifier cannot tell whether a\n    subsequent access to it is valid or not as it\u0027s runtime dependent).\n\n    Signed-off-by: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Acked-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit e7e8078da64ab698730651919345aa13bfef271a\nAuthor: Sasha Levin \u003csasha.levin@oracle.com\u003e\nDate:   Fri Feb 19 13:53:10 2016 -0500\n\n    bpf: grab rcu read lock for bpf_percpu_hash_update\n\n    bpf_percpu_hash_update() expects rcu lock to be held and warns if it\u0027s not,\n    which pointed out a missing rcu read lock.\n\n    Fixes: 15a07b338 (\"bpf: add lookup/update support for per-cpu hash and array maps\")\n    Signed-off-by: Sasha Levin \u003csasha.levin@oracle.com\u003e\n    Acked-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Acked-by: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 94e31c1b8f8b1bc2343fbb22de4feb3842e51316\nAuthor: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\nDate:   Wed Feb 10 16:47:11 2016 +0100\n\n    bpf: fix branch offset adjustment on backjumps after patching ctx expansion\n\n    When ctx access is used, the kernel often needs to expand/rewrite\n    instructions, so after that patching, branch offsets have to be\n    adjusted for both forward and backward jumps in the new eBPF program,\n    but for backward jumps it fails to account the delta. Meaning, for\n    example, if the expansion happens exactly on the insn that sits at\n    the jump target, it doesn\u0027t fix up the back jump offset.\n\n    Analysis on what the check in adjust_branches() is currently doing:\n\n      /* adjust offset of jmps if necessary */\n      if (i \u003c pos \u0026\u0026 i + insn-\u003eoff + 1 \u003e pos)\n        insn-\u003eoff +\u003d delta;\n      else if (i \u003e pos \u0026\u0026 i + insn-\u003eoff + 1 \u003c pos)\n        insn-\u003eoff -\u003d delta;\n\n    First condition (forward jumps):\n\n      Before:                         After:\n\n      insns[0]                        insns[0]\n      insns[1] \u003c--- i/insn            insns[1] \u003c--- i/insn\n      insns[2] \u003c--- pos               insns[P] \u003c--- pos\n      insns[3]                        insns[P]  `------| delta\n      insns[4] \u003c--- target_X          insns[P]   `-----|\n      insns[5]                        insns[3]\n                                      insns[4] \u003c--- target_X\n                                      insns[5]\n\n    First case is if we cross pos-boundary and the jump instruction was\n    before pos. This is handeled correctly. I.e. if i \u003d\u003d pos, then this\n    would mean our jump that we currently check was the patchlet itself\n    that we just injected. Since such patchlets are self-contained and\n    have no awareness of any insns before or after the patched one, the\n    delta is correctly not adjusted. Also, for the second condition in\n    case of i + insn-\u003eoff + 1 \u003d\u003d pos, means we jump to that newly patched\n    instruction, so no offset adjustment are needed. That part is correct.\n\n    Second condition (backward jumps):\n\n      Before:                         After:\n\n      insns[0]                        insns[0]\n      insns[1] \u003c--- target_X          insns[1] \u003c--- target_X\n      insns[2] \u003c--- pos \u003c-- target_Y  insns[P] \u003c--- pos \u003c-- target_Y\n      insns[3]                        insns[P]  `------| delta\n      insns[4] \u003c--- i/insn            insns[P]   `-----|\n      insns[5]                        insns[3]\n                                      insns[4] \u003c--- i/insn\n                                      insns[5]\n\n    Second interesting case is where we cross pos-boundary and the jump\n    instruction was after pos. Backward jump with i \u003d\u003d pos would be\n    impossible and pose a bug somewhere in the patchlet, so the first\n    condition checking i \u003e pos is okay only by itself. However, i +\n    insn-\u003eoff + 1 \u003c pos does not always work as intended to trigger the\n    adjustment. It works when jump targets would be far off where the\n    delta wouldn\u0027t matter. But, for example, where the fixed insn-\u003eoff\n    before pointed to pos (target_Y), it now points to pos + delta, so\n    that additional room needs to be taken into account for the check.\n    This means that i) both tests here need to be adjusted into pos + delta,\n    and ii) for the second condition, the test needs to be \u003c\u003d as pos\n    itself can be a target in the backjump, too.\n\n    Fixes: 9bac3d6d548e (\"bpf: allow extended BPF programs access skb fields\")\n    Signed-off-by: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 27dddc7f1b2fde460e00d29177bcb7d281f65f53\nAuthor: Alexei Starovoitov \u003cast@fb.com\u003e\nDate:   Mon Feb 1 22:39:55 2016 -0800\n\n    bpf: add lookup/update support for per-cpu hash and array maps\n\n    The functions bpf_map_lookup_elem(map, key, value) and\n    bpf_map_update_elem(map, key, value, flags) need to get/set\n    values from all-cpus for per-cpu hash and array maps,\n    so that user space can aggregate/update them as necessary.\n\n    Example of single counter aggregation in user space:\n      unsigned int nr_cpus \u003d sysconf(_SC_NPROCESSORS_CONF);\n      long values[nr_cpus];\n      long value \u003d 0;\n\n      bpf_lookup_elem(fd, key, values);\n      for (i \u003d 0; i \u003c nr_cpus; i++)\n        value +\u003d values[i];\n\n    The user space must provide round_up(value_size, 8) * nr_cpus\n    array to get/set values, since kernel will use \u0027long\u0027 copy\n    of per-cpu values to try to copy good counters atomically.\n    It\u0027s a best-effort, since bpf programs and user space are racing\n    to access the same memory.\n\n    Signed-off-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit b4b194b27c09ae6f258e6856191c99c5c3605fd1\nAuthor: Alexei Starovoitov \u003cast@fb.com\u003e\nDate:   Mon Feb 1 22:39:54 2016 -0800\n\n    bpf: introduce BPF_MAP_TYPE_PERCPU_ARRAY map\n\n    Primary use case is a histogram array of latency\n    where bpf program computes the latency of block requests or other\n    events and stores histogram of latency into array of 64 elements.\n    All cpus are constantly running, so normal increment is not accurate,\n    bpf_xadd causes cache ping-pong and this per-cpu approach allows\n    fastest collision-free counters.\n\n    Signed-off-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit ac39c8564196bef4bf320be4482ae42450714379\nAuthor: Alexei Starovoitov \u003cast@fb.com\u003e\nDate:   Mon Feb 1 22:39:53 2016 -0800\n\n    bpf: introduce BPF_MAP_TYPE_PERCPU_HASH map\n\n    Introduce BPF_MAP_TYPE_PERCPU_HASH map type which is used to do\n    accurate counters without need to use BPF_XADD instruction which turned\n    out to be too costly for high-performance network monitoring.\n    In the typical use case the \u0027key\u0027 is the flow tuple or other long\n    living object that sees a lot of events per second.\n\n    bpf_map_lookup_elem() returns per-cpu area.\n    Example:\n    struct {\n      u32 packets;\n      u32 bytes;\n    } * ptr \u003d bpf_map_lookup_elem(\u0026map, \u0026key);\n    /* ptr points to this_cpu area of the value, so the following\n     * increments will not collide with other cpus\n     */\n    ptr-\u003epackets ++;\n    ptr-\u003ebytes +\u003d skb-\u003elen;\n\n    bpf_update_elem() atomically creates a new element where all per-cpu\n    values are zero initialized and this_cpu value is populated with\n    given \u0027value\u0027.\n    Note that non-per-cpu hash map always allocates new element\n    and then deletes old after rcu grace period to maintain atomicity\n    of update. Per-cpu hash map updates element values in-place.\n\n    Signed-off-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit ec4b45dd051e2a709ae6a361cd204c643c3cf1fc\nAuthor: Alexei Starovoitov \u003calexei.starovoitov@gmail.com\u003e\nDate:   Mon Jan 25 20:59:49 2016 -0800\n\n    perf/bpf: Convert perf_event_array to use struct file\n\n    Robustify refcounting.\n\n    Signed-off-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Signed-off-by: Peter Zijlstra (Intel) \u003cpeterz@infradead.org\u003e\n    Cc: Alexander Shishkin \u003calexander.shishkin@linux.intel.com\u003e\n    Cc: Arnaldo Carvalho de Melo \u003cacme@infradead.org\u003e\n    Cc: Arnaldo Carvalho de Melo \u003cacme@redhat.com\u003e\n    Cc: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Cc: David Ahern \u003cdsahern@gmail.com\u003e\n    Cc: Jiri Olsa \u003cjolsa@kernel.org\u003e\n    Cc: Jiri Olsa \u003cjolsa@redhat.com\u003e\n    Cc: Linus Torvalds \u003ctorvalds@linux-foundation.org\u003e\n    Cc: Namhyung Kim \u003cnamhyung@kernel.org\u003e\n    Cc: Peter Zijlstra \u003cpeterz@infradead.org\u003e\n    Cc: Stephane Eranian \u003ceranian@google.com\u003e\n    Cc: Thomas Gleixner \u003ctglx@linutronix.de\u003e\n    Cc: Vince Weaver \u003cvincent.weaver@maine.edu\u003e\n    Cc: Wang Nan \u003cwangnan0@huawei.com\u003e\n    Cc: vince@deater.net\n    Link: http://lkml.kernel.org/r/20160126045947.GA40151@ast-mbp.thefacebook.com\n    Signed-off-by: Ingo Molnar \u003cmingo@kernel.org\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit a4854ba4e22c0686f055d1937cbb2b6dd0a742f2\nAuthor: Rabin Vincent \u003crabin@rab.in\u003e\nDate:   Tue Jan 12 20:17:08 2016 +0100\n\n    net: bpf: reject invalid shifts\n\n    On ARM64, a BUG() is triggered in the eBPF JIT if a filter with a\n    constant shift that can\u0027t be encoded in the immediate field of the\n    UBFM/SBFM instructions is passed to the JIT.  Since these shifts\n    amounts, which are negative or \u003e\u003d regsize, are invalid, reject them in\n    the eBPF verifier and the classic BPF filter checker, for all\n    architectures.\n\n    Signed-off-by: Rabin Vincent \u003crabin@rab.in\u003e\n    Acked-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Acked-by: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 328be472e45ff8e1ade85989fc644561f445c039\nAuthor: tom.leiming@gmail.com \u003ctom.leiming@gmail.com\u003e\nDate:   Tue Dec 29 22:40:27 2015 +0800\n\n    bpf: hash: use per-bucket spinlock\n\n    Both htab_map_update_elem() and htab_map_delete_elem() can be\n    called from eBPF program, and they may be in kernel hot path,\n    so it isn\u0027t efficient to use a per-hashtable lock in this two\n    helpers.\n\n    The per-hashtable spinlock is used for protecting bucket\u0027s\n    hlist, and per-bucket lock is just enough. This patch converts\n    the per-hashtable lock into per-bucket spinlock, so that\n    contention can be decreased a lot.\n\n    Signed-off-by: Ming Lei \u003ctom.leiming@gmail.com\u003e\n    Acked-by: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 722ab52db1c0b58991ebce641475bed814f98bc3\nAuthor: tom.leiming@gmail.com \u003ctom.leiming@gmail.com\u003e\nDate:   Tue Dec 29 22:40:26 2015 +0800\n\n    bpf: hash: move select_bucket() out of htab\u0027s spinlock\n\n    The spinlock is just used for protecting the per-bucket\n    hlist, so it isn\u0027t needed for selecting bucket.\n\n    Acked-by: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Signed-off-by: Ming Lei \u003ctom.leiming@gmail.com\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 463baae22eae6a38bde5b2696e1d651a3e85e27c\nAuthor: tom.leiming@gmail.com \u003ctom.leiming@gmail.com\u003e\nDate:   Tue Dec 29 22:40:25 2015 +0800\n\n    bpf: hash: use atomic count\n\n    Preparing for removing global per-hashtable lock, so\n    the counter need to be defined as aotmic_t first.\n\n    Acked-by: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Signed-off-by: Ming Lei \u003ctom.leiming@gmail.com\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit ba39f49b13d80157aeffa1bef7b1bdc08420c690\nAuthor: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\nDate:   Thu Dec 17 23:51:54 2015 +0100\n\n    bpf: move clearing of A/X into classic to eBPF migration prologue\n\n    Back in the days where eBPF (or back then \"internal BPF\" ;-\u003e) was not\n    exposed to user space, and only the classic BPF programs internally\n    translated into eBPF programs, we missed the fact that for classic BPF\n    A and X needed to be cleared. It was fixed back then via 83d5b7ef99c9\n    (\"net: filter: initialize A and X registers\"), and thus classic BPF\n    specifics were added to the eBPF interpreter core to work around it.\n\n    This added some confusion for JIT developers later on that take the\n    eBPF interpreter code as an example for deriving their JIT. F.e. in\n    f75298f5c3fe (\"s390/bpf: clear correct BPF accumulator register\"), at\n    least X could leak stack memory. Furthermore, since this is only needed\n    for classic BPF translations and not for eBPF (verifier takes care\n    that read access to regs cannot be done uninitialized), more complexity\n    is added to JITs as they need to determine whether they deal with\n    migrations or native eBPF where they can just omit clearing A/X in\n    their prologue and thus reduce image size a bit, see f.e. cde66c2d88da\n    (\"s390/bpf: Only clear A and X for converted BPF programs\"). In other\n    cases (x86, arm64), A and X is being cleared in the prologue also for\n    eBPF case, which is unnecessary.\n\n    Lets move this into the BPF migration in bpf_convert_filter() where it\n    actually belongs as long as the number of eBPF JITs are still few. It\n    can thus be done generically; allowing us to remove the quirk from\n    __bpf_prog_run() and to slightly reduce JIT image size in case of eBPF,\n    while reducing code duplication on this matter in current(/future) eBPF\n    JITs.\n\n    Signed-off-by: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Acked-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Reviewed-by: Michael Holzheu \u003cholzheu@linux.vnet.ibm.com\u003e\n    Tested-by: Michael Holzheu \u003cholzheu@linux.vnet.ibm.com\u003e\n    Cc: Zi Shen Lim \u003czlim.lnx@gmail.com\u003e\n    Cc: Yang Shi \u003cyang.shi@linaro.org\u003e\n    Acked-by: Yang Shi \u003cyang.shi@linaro.org\u003e\n    Acked-by: Zi Shen Lim \u003czlim.lnx@gmail.com\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 48cfc044b0e191609695dcdb1937296b6f9d0835\nAuthor: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\nDate:   Thu Dec 10 22:33:49 2015 +0100\n\n    bpf, inode: allow for rename and link ops\n\n    Add support for renaming and hard links to the fs. Most of this can be\n    implemented by using simple library operations under the same constraints\n    that we don\u0027t use a reserved name like elsewhere. Linking can be useful\n    to share/manage things like maps across subsystem users. It works within\n    the file system boundary, but is not allowed for directories.\n\n    Symbolic links are explicitly not implemented here, as it can be better\n    done already by doing bind mounts inside bpf fs to set up shared directories\n    f.e. useful when using volumes in docker containers that map a private\n    working directory into /sys/fs/bpf/ which contains itself a bind mounted\n    path from the host\u0027s /sys/fs/bpf/ mount that is shared among multiple\n    containers. For single maps instead of whole directory, hard links can\n    be easily used to do the same.\n\n    Signed-off-by: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Acked-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 72c1982b2f74624362cc72df5b54184c7c7fdff2\nAuthor: Alexei Starovoitov \u003cast@kernel.org\u003e\nDate:   Sun Nov 29 16:59:35 2015 -0800\n\n    bpf: fix allocation warnings in bpf maps and integer overflow\n\n    For large map-\u003evalue_size the user space can trigger memory allocation warnings like:\n    WARNING: CPU: 2 PID: 11122 at mm/page_alloc.c:2989\n    __alloc_pages_nodemask+0x695/0x14e0()\n    Call Trace:\n     [\u003c     inline     \u003e] __dump_stack lib/dump_stack.c:15\n     [\u003cffffffff82743b56\u003e] dump_stack+0x68/0x92 lib/dump_stack.c:50\n     [\u003cffffffff81244ec9\u003e] warn_slowpath_common+0xd9/0x140 kernel/panic.c:460\n     [\u003cffffffff812450f9\u003e] warn_slowpath_null+0x29/0x30 kernel/panic.c:493\n     [\u003c     inline     \u003e] __alloc_pages_slowpath mm/page_alloc.c:2989\n     [\u003cffffffff81554e95\u003e] __alloc_pages_nodemask+0x695/0x14e0 mm/page_alloc.c:3235\n     [\u003cffffffff816188fe\u003e] alloc_pages_current+0xee/0x340 mm/mempolicy.c:2055\n     [\u003c     inline     \u003e] alloc_pages include/linux/gfp.h:451\n     [\u003cffffffff81550706\u003e] alloc_kmem_pages+0x16/0xf0 mm/page_alloc.c:3414\n     [\u003cffffffff815a1c89\u003e] kmalloc_order+0x19/0x60 mm/slab_common.c:1007\n     [\u003cffffffff815a1cef\u003e] kmalloc_order_trace+0x1f/0xa0 mm/slab_common.c:1018\n     [\u003c     inline     \u003e] kmalloc_large include/linux/slab.h:390\n     [\u003cffffffff81627784\u003e] __kmalloc+0x234/0x250 mm/slub.c:3525\n     [\u003c     inline     \u003e] kmalloc include/linux/slab.h:463\n     [\u003c     inline     \u003e] map_update_elem kernel/bpf/syscall.c:288\n     [\u003c     inline     \u003e] SYSC_bpf kernel/bpf/syscall.c:744\n\n    To avoid never succeeding kmalloc with order \u003e\u003d MAX_ORDER check that\n    elem-\u003evalue_size and computed elem_size are within limits for both hash and\n    array type maps.\n    Also add __GFP_NOWARN to kmalloc(value_size | elem_size) to avoid OOM warnings.\n    Note kmalloc(key_size) is highly unlikely to trigger OOM, since key_size \u003c\u003d 512,\n    so keep those kmalloc-s as-is.\n\n    Large value_size can cause integer overflows in elem_size and map.pages\n    formulas, so check for that as well.\n\n    Fixes: aaac3ba95e4c (\"bpf: charge user for creation of BPF maps and programs\")\n    Reported-by: Dmitry Vyukov \u003cdvyukov@google.com\u003e\n    Signed-off-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit c05d78f2deacc2a7c93f1b1c881876953239bc7b\nAuthor: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\nDate:   Mon Nov 30 13:02:56 2015 +0100\n\n    bpf, array: fix heap out-of-bounds access when updating elements\n\n    During own review but also reported by Dmitry\u0027s syzkaller [1] it has been\n    noticed that we trigger a heap out-of-bounds access on eBPF array maps\n    when updating elements. This happens with each map whose map-\u003evalue_size\n    (specified during map creation time) is not multiple of 8 bytes.\n\n    In array_map_alloc(), elem_size is round_up(attr-\u003evalue_size, 8) and\n    used to align array map slots for faster access. However, in function\n    array_map_update_elem(), we update the element as ...\n\n    memcpy(array-\u003evalue + array-\u003eelem_size * index, value, array-\u003eelem_size);\n\n    ... where we access \u0027value\u0027 out-of-bounds, since it was allocated from\n    map_update_elem() from syscall side as kmalloc(map-\u003evalue_size, GFP_USER)\n    and later on copied through copy_from_user(value, uvalue, map-\u003evalue_size).\n    Thus, up to 7 bytes, we can access out-of-bounds.\n\n    Same could happen from within an eBPF program, where in worst case we\n    access beyond an eBPF program\u0027s designated stack.\n\n    Since 1be7f75d1668 (\"bpf: enable non-root eBPF programs\") didn\u0027t hit an\n    official release yet, it only affects priviledged users.\n\n    In case of array_map_lookup_elem(), the verifier prevents eBPF programs\n    from accessing beyond map-\u003evalue_size through check_map_access(). Also\n    from syscall side map_lookup_elem() only copies map-\u003evalue_size back to\n    user, so nothing could leak.\n\n      [1] http://github.com/google/syzkaller\n\n    Fixes: 28fbcfa08d8e (\"bpf: add array type of eBPF maps\")\n    Reported-by: Dmitry Vyukov \u003cdvyukov@google.com\u003e\n    Signed-off-by: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Acked-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 084f062e460c3d681d00b2da803ed5684bda4a49\nAuthor: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\nDate:   Tue Nov 24 21:28:15 2015 +0100\n\n    bpf: fix clearing on persistent program array maps\n\n    Currently, when having map file descriptors pointing to program arrays,\n    there\u0027s still the issue that we unconditionally flush program array\n    contents via bpf_fd_array_map_clear() in bpf_map_release(). This happens\n    when such a file descriptor is released and is independent of the map\u0027s\n    refcount.\n\n    Having this flush independent of the refcount is for a reason: there\n    can be arbitrary complex dependency chains among tail calls, also circular\n    ones (direct or indirect, nesting limit determined during runtime), and\n    we need to make sure that the map drops all references to eBPF programs\n    it holds, so that the map\u0027s refcount can eventually drop to zero and\n    initiate its freeing. Btw, a walk of the whole dependency graph would\n    not be possible for various reasons, one being complexity and another\n    one inconsistency, i.e. new programs can be added to parts of the graph\n    at any time, so there\u0027s no guaranteed consistent state for the time of\n    such a walk.\n\n    Now, the program array pinning itself works, but the issue is that each\n    derived file descriptor on close would nevertheless call unconditionally\n    into bpf_fd_array_map_clear(). Instead, keep track of users and postpone\n    this flush until the last reference to a user is dropped. As this only\n    concerns a subset of references (f.e. a prog array could hold a program\n    that itself has reference on the prog array holding it, etc), we need to\n    track them separately.\n\n    Short analysis on the refcounting: on map creation time usercnt will be\n    one, so there\u0027s no change in behaviour for bpf_map_release(), if unpinned.\n    If we already fail in map_create(), we are immediately freed, and no\n    file descriptor has been made public yet. In bpf_obj_pin_user(), we need\n    to probe for a possible map in bpf_fd_probe_obj() already with a usercnt\n    reference, so before we drop the reference on the fd with fdput().\n    Therefore, if actual pinning fails, we need to drop that reference again\n    in bpf_any_put(), otherwise we keep holding it. When last reference\n    drops on the inode, the bpf_any_put() in bpf_evict_inode() will take\n    care of dropping the usercnt again. In the bpf_obj_get_user() case, the\n    bpf_any_get() will grab a reference on the usercnt, still at a time when\n    we have the reference on the path. Should we later on fail to grab a new\n    file descriptor, bpf_any_put() will drop it, otherwise we hold it until\n    bpf_map_release() time.\n\n    Joint work with Alexei.\n\n    Fixes: b2197755b263 (\"bpf: add support for persistent maps/progs\")\n    Signed-off-by: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Signed-off-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 3fc235b978d41bd1bec76fa9283126fe6801cdca\nAuthor: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\nDate:   Thu Nov 19 11:56:22 2015 +0100\n\n    bpf: add show_fdinfo handler for maps\n\n    Add a handler for show_fdinfo() to be used by the anon-inodes\n    backend for eBPF maps, and dump the map specification there. Not\n    only useful for admins, but also it provides a minimal way to\n    compare specs from ELF vs pinned object.\n\n    Signed-off-by: Daniel Borkmann \u003cdaniel@iogearbox.net\u003e\n    Acked-by: Alexei Starovoitov \u003cast@kernel.org\u003e\n    Acked-by: Hannes Frederic Sowa \u003channes@stressinduktion.org\u003e\n    Signed-off-by: David S. Miller \u003cdavem@davemloft.net\u003e\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 58a5969c25d13339481fbf0c23d449d1c029003d\nAuthor: Anay Wadhera \u003cawadhera@berkeley.edu\u003e\nDate:   Thu May 20 22:00:28 2021 -0700\n\n    Revert \"bpf: fix clearing on persistent program array maps\"\n\n    This reverts commit c9da161c6517ba12154059d3b965c2cbaf16f90f.\n\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit a32aed969feb8bc29832895b8fe6dcf55181b90f\nAuthor: Anay Wadhera \u003cawadhera@berkeley.edu\u003e\nDate:   Thu May 20 22:00:23 2021 -0700\n\n    Revert \"bpf, array: fix heap out-of-bounds access when updating elements\"\n\n    This reverts commit fbca9d2d35c6ef1b323fae75cc9545005ba25097.\n\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 0e181de30ecdb69b72507979a20e0317eb26b28b\nAuthor: Anay Wadhera \u003cawadhera@berkeley.edu\u003e\nDate:   Thu May 20 22:00:16 2021 -0700\n\n    Revert \"bpf: fix allocation warnings in bpf maps and integer overflow\"\n\n    This reverts commit 01b3f52157ff5a47d6d8d796f396a4b34a53c61d.\n\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 55d405ac6e50a5f0a496da252d1fc0053b7efb1a\nAuthor: Anay Wadhera \u003cawadhera@berkeley.edu\u003e\nDate:   Thu May 20 22:00:11 2021 -0700\n\n    Revert \"net: bpf: reject invalid shifts\"\n\n    This reverts commit 35987ff2eaa05d70154c5bd28ebb2b70a7d8368b.\n\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 7e592e9228940029c3fd407503b784f673ec88c5\nAuthor: Anay Wadhera \u003cawadhera@berkeley.edu\u003e\nDate:   Thu May 20 22:00:06 2021 -0700\n\n    Revert \"bpf: fix branch offset adjustment on backjumps after patching ctx expansion\"\n\n    This reverts commit a34f2f9f2034f7984f9529002c6fffe9cb63189d.\n\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 1e454360c63ad9da1afb1f17d016e618068e6fd1\nAuthor: Anay Wadhera \u003cawadhera@berkeley.edu\u003e\nDate:   Thu May 20 22:00:00 2021 -0700\n\n    Revert \"bpf: avoid copying junk bytes in bpf_get_current_comm()\"\n\n    This reverts commit e8e43232627082328fa4016fab1960360360f167.\n\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit e2f52ad516d81c67f897f8ae1e8c221de430a226\nAuthor: Anay Wadhera \u003cawadhera@berkeley.edu\u003e\nDate:   Thu May 20 21:59:51 2021 -0700\n\n    Revert \"bpf/verifier: reject invalid LD_ABS | BPF_DW instruction\"\n\n    This reverts commit 8427d5547d0b63beb70d3858127942f828400ad2.\n\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit b643c8a4050340a1fe14c32c951e5a2d5ae9db85\nAuthor: Anay Wadhera \u003cawadhera@berkeley.edu\u003e\nDate:   Thu May 20 21:58:52 2021 -0700\n\n    Revert \"bpf: fix double-fdput in replace_map_fd_with_map_ptr()\"\n\n    This reverts commit 608d2c3c7a046c222cae2e857cf648a9f89e772b.\n\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit fbb50d299c2311cc6369e77d236d0a0098bff79a\nAuthor: Anay Wadhera \u003cawadhera@berkeley.edu\u003e\nDate:   Thu May 20 21:58:41 2021 -0700\n\n    Revert \"bpf: fix refcnt overflow\"\n\n    This reverts commit 3899251bdb9c2b31fc73d4cc132f52d3710101de.\n\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit ebadfbf36b273986f7273669cab3c6a55f19fcc3\nAuthor: Anay Wadhera \u003cawadhera@berkeley.edu\u003e\nDate:   Thu May 20 21:58:40 2021 -0700\n\n    Revert \"bpf: fix check_map_func_compatibility logic\"\n\n    This reverts commit bb10156f572f06f3b6cadd378e5a0ab3ed8da991.\n\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit c9f2e58ef40b7c33de7777738c14f19e7c627592\nAuthor: Anay Wadhera \u003cawadhera@berkeley.edu\u003e\nDate:   Thu May 20 21:58:12 2021 -0700\n\n    Revert \"bpf: Use mount_nodev not mount_ns to mount the bpf filesystem\"\n\n    This reverts commit 5b7ea922e1754107f77d146011612f2e42600cc1.\n\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit bc301a99c061abc1162c3716a161640479b1fc24\nAuthor: Anay Wadhera \u003cawadhera@berkeley.edu\u003e\nDate:   Thu May 20 21:58:07 2021 -0700\n\n    Revert \"bpf, inode: disallow userns mounts\"\n\n    This reverts commit bfe951d547bf15bf1192abd20773e6603dacadf1.\n\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 8dc982755c60fbcb51a9f50fde8ea14fa59e55a6\nAuthor: Anay Wadhera \u003cawadhera@berkeley.edu\u003e\nDate:   Thu May 20 21:58:02 2021 -0700\n\n    Revert \"bpf: prevent leaking pointer via xadd on unpriviledged\"\n\n    This reverts commit 1a4f13e0a99a85c455ff2f6dc117f6f049c039fa.\n\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 179c39c8c3521e8d8e290b9fcd8858a637872576\nAuthor: Anay Wadhera \u003cawadhera@berkeley.edu\u003e\nDate:   Thu May 20 21:57:55 2021 -0700\n\n    Revert \"bpf/verifier: reject BPF_ALU64|BPF_END\"\n\n    This reverts commit 2ec54b21dd7b25df0f070f1d67db2ea18987e69e.\n\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit df49f6ed8eb48ff19d5d3e6eb725dc498fef3559\nAuthor: Anay Wadhera \u003cawadhera@berkeley.edu\u003e\nDate:   Thu May 20 21:57:48 2021 -0700\n\n    Revert \"bpf: don\u0027t let ldimm64 leak map addresses on unprivileged\"\n\n    This reverts commit 49630dd2e10a3b2fee0cec19feb63f08453b876f.\n\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit f1062c044c662a44f4ff857bd9f61ccbc282063b\nAuthor: Anay Wadhera \u003cawadhera@berkeley.edu\u003e\nDate:   Thu May 20 21:56:46 2021 -0700\n\n    Revert \"bpf: add bpf_patch_insn_single helper\"\n\n    This reverts commit 087a92287dbae61b4ee1e76d7c20c81710109422.\n\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 26702958edcd397a7d8131bf226972a1a44a1e9f\nAuthor: Anay Wadhera \u003cawadhera@berkeley.edu\u003e\nDate:   Thu May 20 21:56:46 2021 -0700\n\n    Revert \"bpf: don\u0027t (ab)use instructions to store state\"\n\n    This reverts commit 0748b80e432584502d1559b1a51b7df58f5e2fce.\n\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 0f77cadba9faa110653a323c6ec76c62ae6cfd0c\nAuthor: Anay Wadhera \u003cawadhera@berkeley.edu\u003e\nDate:   Thu May 20 21:56:45 2021 -0700\n\n    Revert \"bpf: move fixup_bpf_calls() function\"\n\n    This reverts commit 14c7c55f452740549d561e583714b700cd88883e.\n\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit c3f8f34cb2e8cc4822641fdd14b60c1d011ecc35\nAuthor: Anay Wadhera \u003cawadhera@berkeley.edu\u003e\nDate:   Thu May 20 21:56:45 2021 -0700\n\n    Revert \"bpf: refactor fixup_bpf_calls()\"\n\n    This reverts commit 19614eee0644a59a8ea2509a6fbc0e771644a4f2.\n\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 3378c7804d98fc13a46100c545aeeaae9d20637e\nAuthor: Anay Wadhera \u003cawadhera@berkeley.edu\u003e\nDate:   Thu May 20 21:56:44 2021 -0700\n\n    Revert \"bpf: adjust insn_aux_data when patching insns\"\n\n    This reverts commit 648064515d0d91d10d255ab1e3afa3ecffc2943a.\n\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 84df2b434763d55b54b3bcf4f848316b704c2bfd\nAuthor: Anay Wadhera \u003cawadhera@berkeley.edu\u003e\nDate:   Thu May 20 21:56:43 2021 -0700\n\n    Revert \"bpf: prevent out-of-bounds speculation\"\n\n    This reverts commit 9a7fad4c0e215fb1c256fee27c45f9f8bc4364c5.\n\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 57db9aa2122862589c95e7ea29da06b63c5ca633\nAuthor: Anay Wadhera \u003cawadhera@berkeley.edu\u003e\nDate:   Thu May 20 21:56:42 2021 -0700\n\n    Revert \"bpf, array: fix overflow in max_entries and undefined behavior in index_mask\"\n\n    This reverts commit 095b0ba360ff9a86c592c1293602d42a9297e047.\n\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 46b26aef610815060621797cb652974ccd5bb568\nAuthor: Anay Wadhera \u003cawadhera@berkeley.edu\u003e\nDate:   Thu May 20 21:56:29 2021 -0700\n\n    Revert \"bpf: fix branch pruning logic\"\n\n    This reverts commit 1367d854b97493bfb1f3d24cf89ba60cb7f059ea.\n\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 6ad3a678e2b6ddb01239cf2f988f834a3dcff8be\nAuthor: Anay Wadhera \u003cawadhera@berkeley.edu\u003e\nDate:   Thu May 20 21:56:22 2021 -0700\n\n    Revert \"bpf: fix bpf_tail_call() x64 JIT\"\n\n    This reverts commit 361fb0481247bea4da3eb122e685c8b72ef7c8a9.\n\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 8986923ecd1369fbcdd485846546e9466afa6549\nAuthor: Anay Wadhera \u003cawadhera@berkeley.edu\u003e\nDate:   Thu May 20 21:55:49 2021 -0700\n\n    Revert \"bpf: introduce BPF_JIT_ALWAYS_ON config\"\n\n    This reverts commit 28c486744e6de4d882a1d853aa63d99fcba4b7a6.\n\n    Change-Id: Iffebc366a5c2cc47b16e7a09438b018485facb95\n\ncommit 56484cbee4b5f65b9c547556d1a74c35762e57cb\nAuthor: Anay Wadhera \u003cawadhera@berkeley.edu\u003e\nDate:   Thu May 20 21:55:29 2021 -0700\n\n    Revert \"bpf: arsh is not supported in 32 bit alu thus reject it\"\n\n    This reverts commit 7dcda40e52ff0712a2d7d5353c1722cb1f994330.\n\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 672d5d81ded3255b1604724ce2ed95bbc7f9581b\nAuthor: Anay Wadhera \u003cawadhera@berkeley.edu\u003e\nDate:   Thu May 20 21:55:29 2021 -0700\n\n    Revert \"bpf: avoid false sharing of map refcount with max_entries\"\n\n    This reverts commit 96d9b2338bed553c37f759127d8d18c857449ceb.\n\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 61a7684fa21a167a9acbd562515bad6fc3d4234e\nAuthor: Anay Wadhera \u003cawadhera@berkeley.edu\u003e\nDate:   Thu May 20 21:55:28 2021 -0700\n\n    Revert \"bpf: fix divides by zero\"\n\n    This reverts commit b72ba2a0d82447538c7c977ccb3f2b31b19b7767.\n\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 82cb5558180b2ae3563fb50df859cff93867ddc8\nAuthor: Anay Wadhera \u003cawadhera@berkeley.edu\u003e\nDate:   Thu May 20 21:55:27 2021 -0700\n\n    Revert \"bpf: fix 32-bit divide by zero\"\n\n    This reverts commit 02662601a231f8721930168ce71d84bcfb8d9a96.\n\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 1f37f7b70f36c1d382b35b1d3fc1c9154f500555\nAuthor: Anay Wadhera \u003cawadhera@berkeley.edu\u003e\nDate:   Thu May 20 21:55:26 2021 -0700\n\n    Revert \"bpf: reject stores into ctx via st and xadd\"\n\n    This reverts commit faa74a862a9442233bff39a496013a74775fb660.\n\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 2a6fbe491c62059e2973033222c9f6a847ca7c67\nAuthor: Anay Wadhera \u003cawadhera@berkeley.edu\u003e\nDate:   Thu May 20 21:50:08 2021 -0700\n\n    Revert \"bpf: fix incorrect sign extension in check_alu_op()\"\n\n    This reverts commit a6132276ab5dcc38b3299082efeb25b948263adb.\n\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit cdc6d26c1326870560f3564957f0f488a9d18756\nAuthor: Anay Wadhera \u003cawadhera@berkeley.edu\u003e\nDate:   Thu May 20 21:50:03 2021 -0700\n\n    Revert \"bpf: skip unnecessary capability check\"\n\n    This reverts commit c9ea2f8af67399904fe9c72ab5192a0c0ae7f2bf.\n\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit e1a50d45614193600141ac09c8da8be3f015e285\nAuthor: Anay Wadhera \u003cawadhera@berkeley.edu\u003e\nDate:   Thu May 20 21:49:58 2021 -0700\n\n    Revert \"bpf: map_get_next_key to return first key on NULL\"\n\n    This reverts commit ea7c24c78551c8b3e6a7e9824e5ad8ba6224f5fe.\n\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit ae0000ac25fa5071b0c3fb9592fbf7f5a632f042\nAuthor: Anay Wadhera \u003cawadhera@berkeley.edu\u003e\nDate:   Thu May 20 21:49:02 2021 -0700\n\n    Revert \"bpf: fix references to free_bpf_prog_info() in comments\"\n\n    This reverts commit b23dab51e987787e358397b24831505668625b8a.\n\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 674ff13505cc3f05485735f7b1f6913d153dbc80\nAuthor: Anay Wadhera \u003cawadhera@berkeley.edu\u003e\nDate:   Thu May 20 21:48:56 2021 -0700\n\n    Revert \"bpf: generally move prog destruction to RCU deferral\"\n\n    This reverts commit e25dc63aa366fd0f61d1d9ba67b66f5d75fc4372.\n\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 09b6c9c65fa2a549912491d7d69d875e8c326a1d\nAuthor: Anay Wadhera \u003cawadhera@berkeley.edu\u003e\nDate:   Thu May 20 21:48:49 2021 -0700\n\n    Revert \"bpf: support 8-byte metafield access\"\n\n    This reverts commit 3c4bb079e16e222324c68d7594b1ab6f699edfca.\n\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 8ffcd2dc07044fc68ab5badb5eb1e6635147ed82\nAuthor: Anay Wadhera \u003cawadhera@berkeley.edu\u003e\nDate:   Thu May 20 21:48:48 2021 -0700\n\n    Revert \"bpf/verifier: Add spi variable to check_stack_write()\"\n\n    This reverts commit 168cb9b7b2839e861278f9fde03820aba32c4ee0.\n\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 4a7cacf55b2efd9158d8069e9ba223ebf1f86973\nAuthor: Anay Wadhera \u003cawadhera@berkeley.edu\u003e\nDate:   Thu May 20 21:48:47 2021 -0700\n\n    Revert \"bpf/verifier: Pass instruction index to check_mem_access() and check_xadd()\"\n\n    This reverts commit 451624d47005aace4e314b488cb70ba3ee5dcce8.\n\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 413a75b461cabc041d9c057147a919086ce2bb43\nAuthor: Anay Wadhera \u003cawadhera@berkeley.edu\u003e\nDate:   Thu May 20 21:48:46 2021 -0700\n\n    Revert \"bpf: Prevent memory disambiguation attack\"\n\n    This reverts commit 1c74bd22e846b162ea6401e8d43172e0e7256ccf.\n\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 2935a5ca11aa21b53a4afc535a179fa402a85e2e\nAuthor: Anay Wadhera \u003cawadhera@berkeley.edu\u003e\nDate:   Thu May 20 21:48:24 2021 -0700\n\n    Revert \"bpf: silence warning messages in core\"\n\n    This reverts commit 7dd2dc652435c0abb9f05ff9ef0b378fcf743f10.\n\n    Signed-off-by: Chatur27 \u003cjasonbright2709@gmail.com\u003e\n\ncommit 679ee5a4e643ab09c9cedef80cf5e4f61990d120\nAuthor: Stephen Smalley \u003csds@tycho.nsa.gov\u003e\nDate:   Fri Apr 8 13:52:00 2016 -0400\n\n    selinux: distinguish non-init user namespace capability checks\n\n    Distinguish capability checks against a target associated\n    with the init user namespace versus capability checks against\n    a target associated with a non-init user namespace by defining\n    and using separate security classes for the latter.\n\n    This is needed to support e.g. Chrome usage of user namespaces\n    for the Chrome sandbox without needing to allow Chrome to also\n    exercise capabilities on targets in the init user namespace.\n\n    Suggested-by: Dan Walsh \u003cdwalsh@redhat.com\u003e\n    Signed-off-by: Stephen Smalley \u003csds@tycho.nsa.gov\u003e\n    Signed-off-by: Paul Moore \u003cpaul@paul-moore.com\u003e\n    Change-Id: I6b56d3262a73dd8a410785a51e5048aab2c5e254\n\nChange-Id: Iab6c63ba26730279d2699057ea1747d52e2a20fa\n","web_links":[{"name":"GitHub","tooltip":"Open in GitWeb","url":"https://github.com/LineageOS/android_kernel_oneplus_msm8998/commit/c794a655dc883b90f718519667f5a91cda4e9204"}],"resolve_conflicts_web_links":[{"name":"GitHub","tooltip":"Open in GitWeb","url":"https://github.com/LineageOS/android_kernel_oneplus_msm8998/commit/c794a655dc883b90f718519667f5a91cda4e9204"}]},"branch":"refs/heads/lineage-19.0"}},"requirements":[],"submit_records":[]}
